Full text
63,435 characters
· extracted from
preprint-html
· click to expand
Kinetic mechanism for fidelity of CRISPR-Cas9 variants | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Kinetic mechanism for fidelity of CRISPR-Cas9 variants View ORCID Profile Andrew D. Hecht , View ORCID Profile Oleg A. Igoshin doi: https://doi.org/10.1101/2025.01.27.634389 Andrew D. Hecht 1 Department of Bioengineering, Rice University 2 Center for Theoretical Biological Physics, Rice University 3 Rice Synthetic Biology Institute, Rice University Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Andrew D. Hecht Oleg A. Igoshin 1 Department of Bioengineering, Rice University 2 Center for Theoretical Biological Physics, Rice University 3 Rice Synthetic Biology Institute, Rice University 4 Department of BioSciences, Rice University 5 Department of Chemistry, Rice University Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Oleg A. Igoshin For correspondence: igoshin{at}rice.edu Abstract Full Text Info/History Metrics Preview PDF Abstract CRISPR-Cas9 is a nuclease creating DNA breaks at sites with sufficient complementarity to the RNA guide. Notably, Cas9 does not require exact RNA-DNA complementarity and can cleave off-target sequences. Various high-accuracy Cas9 variants have been developed, but the precise mechanism of how these variants achieve higher accuracy remains unclear. Here, we develop a kinetic model of Cas9 substrate selection and cleavage. We parameterize the model using datasets available in the literature, including both high-throughput substrate binding and cleavage data and Förster resonance energy transfer measurements of the Cas9 HNH domain transitions. Based on the observed transition statistics, we predict that the Cas9 substrate recognition and cleavage mechanism must allow for HNH domain transitions independent of substrate binding. Additionally, we show that the enhancement in Cas9 substrate specificity must be due to changes in kinetics rather than changes in substrate binding affinities. Furthermore, the fitted model produces quantitatively realistic cleavage error predictions for substrates with protospacer adjacent motif (PAM)-distal mismatches. Finally, we use our model to identify kinetic parameters for HNH domain transitions that can be perturbed to enable high-accuracy cleavage while maintaining cleavage speeds. Our results refine the biophysical mechanism of Cas9 cleavage to inform future routes for its engineering. I. INTRODUCTION CRISPR-Cas9 is an RNA-guided DNA nuclease derived from bacteria [ 13 ]. Valid DNA target sites for Cas9 require the presence of a short DNA sequence known as a protospacer adjacent motif (PAM) that is recognized via protein-DNA interactions [ 12 ]. After initial PAM binding, the adjacent DNA is unwound to allow the Cas9 guide RNA to interrogate the local sequence, causing the formation of an RNA-DNA hybrid known as an R-loop. The Cas9 RuvC and HNH nuclease domains then become catalytically active following full R-loop formation and can cleave the backbone of both DNA strands. Cas9 is capable of cleaving at target sites even in the presence of DNA-RNA mismatches [ 5 , 11 ]. Generally, mismatches at the PAM-distal end of the target are better tolerated than PAM-proximal (so-called “seed region”) mismatches. Inappropriate cleavage at such off-target sites can cause DNA base insertion/deletion (indel) mutations as well as genomic-scale damage in the form of large deletions [ 2 ], which both present significant safety concerns for clinical applications of Cas9-based technologies. Accordingly, multiple efforts have been made to develop high-fidelity CRISPR-Cas9 platforms [ 1 , 6 , 17 , 28 , 29 ]. The high-fidelity variants developed by Slaymaker and Kleinstiver were built under the “excess energy” hypothesis [ 17 , 28 ]. Under this hypothesis, the Cas9-DNA binding interaction is thought to be more energetically favorable than necessary to achieve catalytic activation and subsequent DNA cleavage. Thus, the bound complex would be destabilized by removing Cas9 residues that interact non-specifically with the DNA backbone via electrostatics. This destabilization then increases the probability that RNA-DNA mismatches would disrupt enzymatic activation. The resulting variants were indeed shown to achieve higher substrate selection fidelity than the wild-type Cas9 variant [ 17 , 28 ]. However, there was no significant difference in substrate binding affinity compared to WT-Cas9 when bound to either fully matched or partially mismatched sequences [ 1 ]. Therefore, an alternative mechanism could be required to explain the improved specificity. In an effort to explain the specificity enhancement observed in the high-fidelity variants, single-molecule Förster resonance energy transfer (smFRET) experiments examined the effects of the dynamics of the Cas9 HNH domain on substrate selection specificity [ 1 , 3 ]. It was found that the Cas9 HNH domain can adopt distinct conformational states, including an intermediate FRET “checkpoint” conformation that appears to play a role in rejected distally-mismatched substrates. The HNH domain dynamics of the high-fidelity Cas9 variants were found to be perturbed relative to the wild-type variant, with the high-fidelity variants being significantly more sensitive to the presence of PAM-distal mismatches. However, it remains unclear whether the dynamics of the HNH domain directly control cleavage specificity or if the changes in the dynamics are a result of some other process that underlies the change in specificity (e.g. R-loop formation and collapse). In order to develop our understanding of Cas9 substrate selection, we developed a biophysical model of target site selection and cleavage. Our model proposes a kinetic scheme to model Cas9 transitions with parameters estimated based on fitting of multiple datasets in the literature, including single-molecule FRET measurements of the Cas9 HNH domain [ 1 , 3 ] and high-throughput measurements of Cas9-DNA binding and cleavage [ 14 ]. We use the parameterized model to examine the relative importance of key Cas9 target selection processes, including R-loop formation and HNH domain movements, toward off-target rejection. Finally, we use our model to examine trade-offs between on-target cleavage speed and off-target specificity and show that there are few intrinsic trade-offs preventing the development of high-speed, high-specificity Cas9 variants. II. RESULTS A. Changes in binding are insufficient to explain high-fidelity Cas9 accuracy One of the simplest ways to model CRISPR-Cas9 substrate selection is as a linear (1D) process ( Fig. 1A ) with two possible branches corresponding to cleaving either a correct, on-target substrate (R) or an incorrect, off-target substrate (W). Such a model is a special case of the Shvets-Kolomeisky model of Cas9 [ 25 ] where there is only a single off-target site available. Binding to either substrate is a two-step process, starting with initial binding to the PAM site, which is then followed by the formation of a full R-loop. Once the R-loop is fully formed, Cas9 can cleave the bound substrate to form a cleaved DNA product. The substrate selection error η can then be defined as a ratio of probabilities η = π W / π R , where π R/W are the conditional probabilities of cleaving one substrate before the other, given that the system starts in state S 0,1 at the initial time t = 0. Download figure Open in new tab Figure 1. Cas9 selection between an on-target and off-target substrate can be represented by a simple model with two bound states for each substrate (A). The bound states correspond to an initial PAM-bound state and a fully bound, R-loop state. Amino acid substitutions, such as those involved in developing high-fidelity Cas9 variants, can affect Cas9 function by modulating the underlying free-energy landscape (B). The WT-Cas9 free-energy landscape (black) could be perturbed to yield the high-fidelity landscape (red) by changing either the stabilities of individual states (i.e., binding energies) or the barriers between states, which correspond to activation energies. Computing the cleavage error for such a model we see that changes in the stability of states, such as destabilizing the off-target PAM state (C) or the R-loop state (D) are insufficient to affect Cas9 specificity. Only changes in transition state activation energies, such as that for cleaving the off-target substrate (E), can affect specificity. Thermodynamically and kinetically, this model type can be represented by a free energy landscape ( Fig. 1B ) in which the individual states of the model correspond to potential wells. Adjacent states are separated by free-energy barriers, with maxima representing transition state energies for transitions between states. In this landscape, two types of perturbations are possible: changes in the energies of the stable intermediate states ( ϵ i ) or changes in the transition state energies . Previous theoretical work has shown that only perturbations in transition state energies can affect the error of biological substrate selection processes; in other words, the error is under kinetic control [ 19 ]. The result holds for the model of Cas9 in Fig. 1A : our calculations demonstrate that decreasing the energy of the off-target R-loop or PAM-bound states relative to the on-target substrate (i.e., as increasing their relative stability) does not change the error ( Fig. 1C,D ). In contrast, if the substrate cleavage transition-state energy for the off-target substrate is increased, the error decreases substantially ( Fig. 1E ). Therefore, we conclude that even if the high-fidelity mutations affected the Cas9-DNA binding affinity, it would not cause any change in substrate selection accuracy. B. HNH domain transition statistics imply that the energy landscape is multi-dimensional Experimental single-molecule FRET studies of Cas9 identified three distinct configurational states corresponding to low, mid, and high FRET conformations [ 1 , 3 ]. Each conformation represents the Cas9 HNH domain moving from the inactive (low-FRET) state towards the catalytically active (high-FRET) state ( Fig. 2A ). Download figure Open in new tab Figure 2. The Cas9 HNH domain was identified as a potential contributor to cleavage specificity and found in smFRET studies to have three possible conformational states corresponding to low-, mid-, and high-FRET configurations (A). The HNH domain dynamics can yield insights into the structure of the Cas9 kinetic network. The statistics of HNH domain transition paths (B) which are the time required for the HNH domain to transit from the low-FRET state to the high-FRET state (green path) can help differentiate between possible model structures. The smFRET dataset suggests that the distribution of transition paths is broad, with CVs greater than unity for both WT and HF1-Cas9 variants (C). The broad distribution of transition paths suggests that the underlying kinetic network must be multi-dimensional, with a 1D kinetic model (boxed network in Fig. 1A ) being incapable of producing transition path distributions with CVs greater than one (C, hatched bars). Additionally, the distribution of residence times for the three HNH domain states show significant deviations ( p ≪ 0.05, Kolmogorov-Smirnov test) from reference exponential distributions (dashed reference lines, passing through the 1st and 3rd quartiles of the sample data), which we expect only for multi-dimensional models (D). Recent theoretical results have shown that the statistics of single-molecule transitions can reveal deep insights into the underlying thermodynamic landscape. In particular, the statistics of transition path times can help differentiate between different kinetic model structures [ 24 ]. Here, a transition path occurs when Cas9 in the low-FRET HNH domain successfully transits through the mid-FRET state to the high-FRET state ( Fig. 2B ). We identified all such transition paths in the smFRET dataset for both WT-Cas9 and HF1-Cas9. For each variant, the distribution of transition path times was broad, with coefficients of variation (CVs) greater than one ( Fig. 2C ). The CV of a sample is defined as the standard deviation ( σ ) scaled by the mean ( µ ), CV = σ / µ . When the CV is greater than one, it indicates that the underlying thermodynamic landscape must be multi-dimensional [ 24 ]; in essence, multiple paths must exist for the system to transit from point A to point B. In the context of Cas9, multi-dimensionality implies that the simple 1D model is insufficient to properly model Cas9 behavior. As expected, Monte Carlo simulations of the 1D model parameterized with rate constants from Liu et al. [ 18 ] produce transition path CVs less than or equal to one most of the time for both WT-Cas9 and HF1-Cas9 ( Fig. 2C ). We can also examine the residence times distribution for each HNH domain state to gain insight into the proper model structure. The built-in MATLAB ® probplot function generates probability plots [ 32 ] that allow sample data to be compared against the quantiles (converted to cumulative probabilities) of a reference probability distribution. If Cas9 binding and cleavage were a 1D process, as in the boxed network in Figure 1A , the distribution of residence times for the low and high-FRET states would be exponential. Thus, the sample data would be roughly linear when plotted against quantiles from exponential distributions with identical means (2D). However, the experimental data shows substantial deviations from the exponential distribution for all three HNH domain states, which further suggests that the underlying kinetic network must be multi-dimensional. C. A 2D kinetic model can match experimental Cas9 datasets The simplest multi-dimensional kinetic network that can describe Cas9 behaviors is a 2D network in which target recognition (PAM-binding, R-loop formation/collapse) can occur before or after the HNH domain transitions ( Fig. 3 ). We implement this model similarly to the 1D model, with Cas9 selecting between an on-target and off-target substrate. Unbound Cas9 can exist in either the low-FRET or mid-FRET state, as HNH domain transitions were detected when Cas9 is only bound to the guide-RNA [ 1 , 3 ]. Download figure Open in new tab Figure 3. A 2D model of Cas9 target binding and cleavage can be constructed by allowing the R-loop formation/collapse dynamics to occur orthogonally to the movements of the HNH domain. Reactions can be grouped into three distinct sets of processes, including those involved in target recognition (green), conformational proofreading (red), and substrate cleavage (blue). The individual model states are binned into low-FRET mid-FRET and high-FRET states. We fitted the 2D model to various datasets, including the HNH domain smFRET dataset and a high-throughput dataset of Cas9 substrate cleavage rates and binding affinities. The fitted model is high quality, with almost all on-target data points fitted within ±25% margin ( Fig. 4A ). Download figure Open in new tab Figure 4. The 2D model can be well fitted to a dataset consisting of high-throughput binding and cleavage measurements as well as HNH domain transition statistics derived from smFRET (A). The fitted model produces off-target cleavage error predictions that are qualitatively consistent with experimental measurements in both in-vitro and in-vivo settings (B). Monte Carlo simulations of the fitted model produce transition path distributions consistent with the experimental dataset (C). Further, Monte Carlo simulations with the fitted 2D model produce non-exponential HNH domain residence time distributions that deviate from the reference exponential distribution (dashed reference lines, passing through the 1st and 3rd quartiles of the sample data), which are consistent with the experimental smFRET dataset (D). The model produces cleavage error predictions that are qualitatively consistent with both in-vivo and in-vitro datasets ( Fig. 4B ). Quantitative differences between the experimental datasets and the model predictions are likely due to limitations in the available fitting data. Specifically, the model is fitted to in-vitro data. Thus, the data does not account for cellular factors that could affect Cas9 specificity, such as the presence of histones or active DNA remodeling. Furthermore, due to the low throughput of the FRET experiments, we were required to assume that the dynamics of the HNH domain would depend solely on the number of PAM-distal mismatches; thus, the N mismatch substrates in the data set were fitted to have the same HNH domain dynamics as the corresponding substrate in the smFRET data set. We additionally performed Monte Carlo simulations with the fitted model to check for agreement with the smFRET dataset; the Monte Carlo simulations were performed for the single-substrate case (no off-target substrate present, . The simulations produce transition path CVs that are greater than one for WT-Cas9, which is consistent with the experimental data ( Fig. 4C ). The HF1-Cas9 transition path CV is less than one, but still remains within the 95% confidence interval obtained from the experimental dataset. Monte Carlo simulations with the fitted model produce transition path CVs that are greater than one, which is consistent with experimental data. The simulations of the WT-Cas9 model also produce residence-time distributions that are non-exponential (for the low-FRET state, p ≈ 10 −4 , Kolmogorov-Smirnov test), which is consistent with what is observed in the experimental data ( Figs. 2D and 4D ). D. HNH domain dynamics are a key contributor to specificity Next, we attempted to use our model to understand the relative importance of key Cas9 processes for substrate selection specificity. Previous work has suggested that processes such as the dynamics of the HNH domain [ 1 , 3 ], R-loop formation and collapse dynamics [ 8 , 27 ], and the rate of substrate cleavage [ 18 ] all contribute to the accuracy of substrate selection. Each of these processes is represented as distinct transitions in our model, which allows us to gauge their relative importance quantitatively. We performed a global parameter swapping routine to determine which processes in the HF1 high-fidelity Cas9 model led to the greatest improvement in the substrate selection error for wild-type Cas9. We focused on the HNH domain dynamics (reactions 3, 4, 5, 6 in Fig. 3A ) and the R-loop formation and collapse dynamics (reactions 2, 7, and 8 in 3). The microscopic substrate cleavage rate was assumed to be identical across both Cas9 variants and substrates and, therefore, is not relevant to this analysis. The native WT-Cas9 and HF1-Cas9 (ensemble average) cleavage error predictions for the distal-mismatch substrates are shown in blue and purple in Fig. 5 ; the trend is qualitatively consistent with experimental measurements of Cas9 off-target accuracy [ 15 ]. We see that while both the HNH domain dynamics (yellow) and R-loop dynamics (orange) lead to a reduction in WT-Cas9 cleavage error, the HNH domain dynamics have a larger effect ( Fig. 5A ). From this, we conclude that the HNH domain dynamics are the primary contributor to specificity for distally mismatched substrates. Download figure Open in new tab Figure 5. Globally swapping sets of parameters corresponding to R-loop formation/collapse as well as HNH domain transitions between the WT and HF1-Cas9 models can yield insights into the relative importance of each process on specificity. The swapped models suggest that the HF1-Cas9 HNH domain dynamics (A, yellow bars) have a greater impact on specificity than HF1-Cas9 R-loop formation/collapse dynamics (A, orange bars). Additionally, the low-FRET R-loop formation/collapse rates and the HNH domain transition rates in the R-loop state are predicted to have the greatest effect on on-target cleavage speed (B). We additionally performed a local sensitivity analysis to identify which model parameters are predicted to have the greatest effect on the on-target cleavage speed, V = 1/ τ . For both the WT and HF1 variants, the R-loop formation and collapse rates ( k 2,−2 ) and HNH domain transitions in the R-loop state ( k 3,−3 , k 4,−4 ) have the most significant effect on the speed of on-target cleavage ( Fig. 5B ). E. There are no intrinsic speed-specificity trade-offs for Cas9 Finally, we used our model to examine speed-specificity trade-offs in Cas9. Previous experimental studies identified a clear trade-off in on-target cleavage efficiency (measured as on-target indel formation rate) and off-target cleavage specificity (measured as the off-target indel formation rate relative to the on-target) [ 15 ]. Our model can be used to compute analogous quantities to gauge the effects of distinct thermodynamic parameter perturbations on Cas9 specificity and efficiency. The model can compute the off-target cleavage specificity directly, however we must use the speed of on-target cleavage as a proxy for on-target cleavage efficiency. While the model predicts the existence of a speed-specificity trade-off when the barrier for substrate cleavage is perturbed ( Fig. 6A ), there is no such trade-off observed for perturbations in the R-loop formation barrier ( Fig. 6B ) or for perturbations in the HNH domain transition barriers in the fully formed R-loop state ( Fig. 6C,D ). Overall, the lack of a clear speed-specificity trade-off for the HNH domain dynamics or R-loop formation and collapse dynamics suggests no biophysical limitation to developing high-specificity, high-efficiency Cas9 variants. ‘ Download figure Open in new tab Figure 6. Perturbing free-energy barriers in the fitted model allows us to identify potential parameter regimes that would allow Cas9 to achieve both high on-target cleavage efficiency and low off-target cleavage (high-specificity). While there is a clear trade-off for perturbations in the substrate cleavage barrier (A), for R-loop formation/collapse (B) as well as HNH domain transitions in the fully formed R-loop state (C, D) there is no appreciable trade-off, with Cas9 being able to increase on-target cleavage speed with minimal impacts on cleavage specificity. III. DISCUSSION Here, we have presented a model of CRISPR-Cas9 substrate selection and cleavage. The model fits diverse experimental data sources, including high-throughput measurements of Cas9-mediated DNA binding and cleavage [ 14 ], as well as single-molecule measurements of the Cas9 HNH domain dynamics when bound to PAM-distally mismatched DNA substrates [ 1 , 3 ]. The fitted model reproduces trends observed in the literature, such as the effective mid to high-FRET transition rate, which was noted to be higher in WT-Cas9 than the HF1-Cas9 variant [ 1 ]. The HF1-Cas9 [ 17 ] high-fidelity Cas9 variant was developed under the “excess energy” hypothesis, which requires that the Cas9-DNA binding affinity be decreased to improve the accuracy of substrate selection. However, we have shown that changes in binding affinities are insufficient to affect the accuracy of Cas9 substrate selection, which is consistent with both experimental measurements [ 1 ] and prior theoretical results for generic reaction networks [ 20 ]. Instead, any change in the accuracy of Cas9 requires changes in the transition state energies (i.e. kinetics) of substrate binding and cleavage. Our model predicts that the HF1-Cas9 variant is more likely to get stuck in the mid-FRET, checkpoint state compared to the wild-type variant, as the dissociation constant for entering the checkpoint state with a fully-formed R-loop ( K D = k −3 / k 3 ) is much higher than for wild-type Cas9. Additionally, when HF1-Cas9 is targeted to off-target substrates with a single distal mismatch, the enzyme tends to get stuck in the low-FRET, full R-loop state ( S 2,1 ) compared to WT-Cas9; the fraction of Cas9 proteins in either of the two mid-FRET states are predicted to be roughly equal between WT-Cas9 and HF1-Cas9. A previous model developed by Klein and Eslami-Mossallam [ 4 , 16 ] considers Cas9 substrate binding and cleavage as a 1D birth-death process in which the R-loop grows a single base at a time until fully formed. Under this model, any HNH domain movements would be required to occur either simultaneously with R-loop formation or as an additional step following the formation of the full R-loop. In contrast, statistical analysis of the Cas9 HNH domain dynamics (from smFRET) has revealed a broad distribution of HNH domain transition path times, which suggests that the HNH domain dynamics are best modeled as processes independent of R-loop formation and collapse. Overall, this implies that the HNH domain is not tightly mechanically coupled to the formation of the R-loop, and should instead be modeled as separate, distinct reaction steps that can occur in the presence of an incomplete or partially formed R-loop. Structural studies by Pacesa et al. show that the HNH domain requires partial R-loop formation (at least through the “seed” region) to undock [ 22 ]. In our model, because we allow for differences in PAM unbinding rates between the on and off-target substrates, our PAM-bound state is more akin to an early (or “seed” region) R-loop state. In this case, we expect the HNH domain to still be capable of movement in either of the bound states. Regardless, the observation of transitions between the low-FRET and mid-FRET states in the absence of DNA substrate suggests that partial R-loop formation is not a strict requirement to observe HNH domain movements [ 3 ]. Additionally, separate smFRET studies by Singh et al. found that R-loop formation occurs as a two-step process with an initial “sampling” state followed by the formation of a fully constituted R-loop [ 26 ]. The observation that R-loop formation occurs as a two-step process is consistent with our model as we explicitly require binding to occur in two steps. Our fitted model also suggests that HF1-Cas9 is more likely to reject off-target substrates from the initially bound, partial R-loop state than wild-type Cas9. This appears to be consistent with later findings by Singh et al., which showed that the R-loop formation and collapse dynamics of HF1-Cas9 are more sensitive to the presence of PAM-distal mismatches [ 27 ]. Our model produces cleavage error predictions for PAM-distally mismatched substrates that are qualitatively consistent with both in-vivo and in-vitro measurements of Cas9 cleavage error. However, some quantitative differences remain between model predictions and experimental measurements. Namely, our model predicts cleavage errors that are intermediate between the in-vivo and in-vitro measurements for a given number of distal mismatches. There are multiple possible causes for this. First, due to limited smFRET data (which is only available for five substrates, and is missing for the wild-type 1bp and 2bp mismatch substrates), we were forced to assume that only the number of PAM-distal mismatches would affect the HNH domain dynamics. Therefore, for each substrate in the high-throughput dataset, we assumed that the HNH domain dynamics to be identical to the corresponding smFRET substrate based on the number of distal mismatches. In reality, the precise identity and ordering of PAM-distal mismatches, as well as the presence of PAM-proximal mismatches, would likely affect the behavior of the HNH domain. Further, even with the wild-type 1bp and 2bp substrates displaying steady-state FRET histograms that are largely similar to the on-target substrate [ 1 ], it is possible that the precise HNH domain dynamics vary for these conditions and could affect the fitted transition rates. Second, the fitting data we use to parameterize the model comes exclusively from in-vitro experiments. In the in-vitro context, multiple factors that could affect Cas9 cleavage are absent, such as the presence of histones in nucleosomes [ 10 ] and DNA stretching due to active processes such as RNA transcription and nucleosome remodeling [ 21 ]. The influence of these factors could affect the accuracy of Cas9 and should be accounted for in future models, especially as they pertain to genome therapies. Importantly, our model reveals the relative importance of different Cas9 biophysical processes in substrate selection. Prior work has focused on the HNH domain dynamics [ 1 , 3 ], R-loop formation and collapse dynamics [ 8 , 27 ], and substrate cleavage rate [ 18 ] as possible contributors to substrate specificity. Our model suggests that while both the HNH domain dynamics and R-loop formation/collapse dynamics are important for substrate selection, the dynamics of the HNH domain are the most significant factor controlling off-target cleavage for PAM-distally mismatched substrates. Our work shed important insights into trade-offs in Cas9 engineering. Previous work by Kim et al. has identified trade-offs between Cas9 substrate selection fidelity and on-target cleavage efficiency [ 15 ], in which the efficiency of on-target cleavage tends to decrease as the ability to reject off-target substrates increases. Examining different perturbations in our model suggests that it is generally possible for Cas9 to avoid this trade-off. While our model cannot predict on-target cleavage efficiency as defined by Kim et al., instead predicting the speed of on-target cleavage, it suggests that further engineering should allow for high-efficiency, high-specificity Cas9 variants. Indeed, recent work using generative artificial intelligence (GenAI) models has lead to the development of an artificial Cas9-like enzyme (OpenCRISPR-1) that appears to achieve greater off-target specificity than wild-type Cas9 while maintaining wild-type-like on-target cleavage efficiency [ 23 ]. The development of the OpenCRISPR-1 variant is consistent with our finding that there is no intrinsic barrier preventing high-efficiency, high-specificity Cas9 variants. However, it will still be important to test the OpenCRISPR-1 variant on a much larger set of off-target substrates to get a more accurate sense of the specificity enhancement relative to wild-type Cas9. IV. METHODS A. Single-Molecule FRET Data and Processing Single-molecule FRET traces for wild-type, HF1, and eSpy(1.1)-Cas9 were obtained from the study of Chen et al. 2017 [ 1 , 3 ]. Chen et al. acquired smFRET data for each Cas9 variant bound to either a fully matched, on-target substrate or PAM-distally mismatched substrates with one to four mismatches. Each trace in the dataset consists of the Cy3 (donor) and Cy5 (acceptor) signal intensity (in a.u.) over time for an individual Cas9-DNA particle with FRET labels attached to the Cas9 HNH and REC1 domains [ 1 ]. The Chen dataset was acquired using a camera with a 10Hz acquisition frequency [ 1 ]; therefore, the time resolution of the data is limited to 100 milliseconds. We were unable to obtain FRET traces corresponding to the WT-Cas9 1bp and 2bp mismatch data; we make the assumption that the WT HNH domain dynamics for the 1bp and 2bp mismatch cases will be similar to the fully-matched, on-target case because they have similar steady-state FRET histograms [ 1 ]. Therefore, we re-use the on-target FRET data for the WT-Cas9 1bp and 2bp conditions. The FRET traces were preprocessed prior to statistical analysis. First, the donor and acceptor signal intensity traces were smoothed using a 3rd-order Savitsky-Golay filter with a window size of 11 timesteps. The MATLAB ® function sgolayfilt in the Signal Processing Toolbox was used to implement the filter. The timepoint corresponding to acceptor photobleaching was identified using the photobleach index function in the ebFRET package [ 30 , 31 ]; individual traces were then truncated at the point of photobleaching. The dataset was then filtered to select traces displaying anti-correlated donor and acceptor signals (Pearson’s ρ < −0.5), which is indicative of a true FRET signal. Traces shorter than 50 frames (5 seconds) after truncation were discarded. For each trace, the effective FRET signal was computed as the ratio where D and A are the donor and acceptor intensities, respectively. B. FRET State Discretization The processed smFRET traces were discretized using the empirical Bayes hidden Markov model (HMM) tool ebFRET [ 30 , 31 ]. The ebFRET tool implements a linear HMM that can be fitted simultaneously to large datasets by learning a distribution of model parameters to account for variability between molecules due to slight differences in experimental conditions, photophysical effects, and other factors [ 30 ]. Discretizing using an HMM helps to assign states to each point in the smFRET trace while minimizing the effects of signal noise. We fitted the traces in ebFRET to a six-state HMM with default settings. Prior to fitting, outlier FRET intensities within each trace were clipped to the range [−0.2, 1.2]; traces with more than 10 outliers were discarded. The discretized states were then assigned to the low-FRET (inactive), mid-FRET (checkpoint), and high-FRET (active) HNH domain states by a simple thresholding procedure with thresholds set based on the mean FRET efficiencies reported by Chen et al. [ 1 ]. The threshold T ij between adjacent states i and j was set as the midpoint between the FRET efficiencies: C. HNH Domain Residence Time Statistics HNH domain state residence times were computed from the FRET traces using custom scripts in MATLAB ® . After each detected transition, the time until the next transition was computed and taken to be the dwell time τ . HNH domain state residence times were computed from each discretized trace in the smFRET dataset. The mean residence time for each FRET state i is an average where each τ j represents the time the system stays in state i prior to a detected state transition (residence time), and N tot is the total number of detected events. For each trace, we discard the first and last detected transitions as we do not know the proper transition start and end times for these transitions. D. HNH Domain Steady-State Occupancy The probability of observing each HNH domain state at steady-state was calculated from the smFRET dataset as an average over the final timepoint for each trace. We assume that each trace is sufficiently long for the ensemble of traces to achieve a steady state. The steady-state probabilities are computed as where n is the total number of traces and N i is the number of traces that end in state i . E. HNH Domain Transition Paths Transition paths are computed from smFRET traces by tabulating all transitions from the mid-FRET (checkpoint state) into the high-FRET (active state). The initial transition path for a given FRET trace is discarded if the trace starts in the mid-FRET state, as there is no way to determine how long the molecule has resided in the mid-FRET state. After a transition occurs, the FRET trace must return to the low-FRET state before additional transition paths are counted. F. High-Throughput Data Preprocessing High-throughput measurements of WT-Cas and HF1-Cas9 substrate binding and cleavage were obtained from Jones et al. [ 14 ]. For each substrate in the dataset, we computed the effective cleavage mean first-passage time from the measured cleavage rate k i as The effective cleavage error for each substrate can then be computed as a ratio of mean first-passage times where τ 0 is the cleavage mean-first passage time for the on-target substrate. The dataset was filtered to select substrates that possess a valid NGG PAM sequence, as well as substrates for which the effective cleavage error η i is less than one. The high-throughput dataset was preprocessed using a custom script in Python 3. G. Transition State Theory The transition rate constant for each transition i → j can be computed via transition state theory as where E i is the free-energy of state i and is the free-energy barrier for the i → j transition. The parameter ω is a pre-exponential frequency factor and β = 1/( K B T ), where K B is the Boltzmann constant. H. Forward Master Equation The forward chemical master equation is defined by the system of differential equations where is a vector of state probabilities and the matrix K encodes the state transitions. The system of equations is subject to the normalization constraint which ensures that each P i can be interpreted as a probability. We can obtain the steady-state probability for each system state with the condition I. Backward Master Equation The backward chemical master equation is defined as a system of differential equations where each F i ( t ) is a probability distribution giving the probability of reaching some terminal state (e.g.: the cleaved substrate) at time t , conditioned on the system starting in state i at time t = 0. The matrix K is a transition matrix that encodes the state transitions of the model, and is a vector of source terms that encodes the initial conditions (e.g., F q (0) = 1 if state q is a terminal state). Note that the transition matrix K is distinct from the transition matrix used in the forward master equation. The system of differential equations defining the backward master equation can be transformed into Laplace space to yield which can be solved algebraically for each . Critical properties of the system can then be obtained from , which represents the system starting in the unbound, low-FRET state. The system splitting probabilities π R and π W can be obtained by taking the limit of as the Laplace variable s goes to zero: The effective substrate selection error (cleavage error) can then be defined as a ratio of splitting probabilities: The backward master equation can additionally used to obtain information on the system dynamics, such as the mean first-passage time (MFPT, τ ) to on-target (R) substrate cleavage. The MFPT is obtained by computing We can also compute the effective cleavage speed as the inverse of the τ , J. Dissociation Constant and Binding Affinities The apparent Cas9-substrate dissociation constant ( K D ) can be computed from our kinetic modeling framework by computing the steady-state bound fraction, P bound for varying substrate concentrations ( x ) using the forward master equation [ 14 ]. The bound fraction is defined as the sum of all bound states or equivalently where P 0,1 and P 0,2 are the low-FRET and mid-FRET unbound states. The dissociation constant K D is equivalent to the concentration of substrate at which the available Cas9-gRNA is half-saturated, i.e. P bound = 1/2. We can obtain as a function of model parameters by solving the equation for the concentration x = K D , where is a vector of model parameters. The normalized binding affinity (Δ ABA ) for off-target substrates is computed numerically as the ratio of the effective dissociation constants K D,R and K D,W , i.e. Δ ABA = K D,W / K D,R . K. HNH Domain Residence Time Statistics We use the framework of Gopich and Szabo [ 9 ] to compute residence time distributions for each of the HNH domain FRET states. For each HNH domain FRET state, we consider only the DNA bound states (i.e. ignoring states and ). The matrix V encodes the transitions that enter the meta-state of interest (e.g. ‘low-FRET state’). The entry probability is computed from the steady-state solution of the system as The matrix U encodes the transitions that exit the meta-state of interest. We can solve for the generating function for the distribution of exit times by solving where is the generating function in Laplace space. The raw moments of the residence time distribution can be computed from bydifferentiating and taking the Laplace variable s → 0. The residence time mean is identical to the first raw moment: The residence time variance is obtained as from which we can then obtain the residence time CV: L. Model Optimization We parameterized the model using a two-stage fitting procedure. First, we fitted the model to the WT-Cas9 and HF1-Cas9 on-target data along with data from a 4bp mismatched substrate. After the initial fitting, we then fixed the on-target parameters and fitted the model to the remaining off-target substrates in the dataset individually. For both fitting stages, we used the built-in particle swarm optimizer ( particleswarm ) available in MATLAB ® Global Optimization Toolbox. All model fitting runs were performed on the Rice University NOTS high-performance computing cluster in MATLAB ® R2021a. Additional details on the model implementation and fitting procedure are given in the Supplemental Information. M. Sensitivity Analysis Local sensitivity analysis is performed by computing logarithmic gains as where x is a model parameter of interest and f ( x ) is some model output. The terms x 0 and f ( x 0 are the initial parameter value and the model response at that point. N. Monte Carlo Simulation Stochastic simulations of the model were performed using a custom implementation of the Gillespie algorithm [ 7 ] in MATLAB ® R2021a. Simulations were performed for N = 1000 trajectories, with each trajectory simulated for t = 100 seconds. At the beginning of each trajectory we simulated an additional 800 seconds of burn-in time to ensure that the simulation results were unbiased by the initial conditions. O. Cleavage Specificity The average cleavage specificity across all off-target substrates in the dataset for each variant is computed from the average cleavage error as P. Probability Plots Probability plots comparing a data sample against a theoretical reference distribution are generated using the built-in MATLAB ® function probplot . The probplot function generates a probability plot in which the quantiles from the theoretical distribution are rescaled and converted to the equivalent cumulative probability. The reference line generated by probplot is generated from the theoretical distribution and is set to pass through the 1st (25%) and 3rd (75%) quartiles of the sample data. Q. Statistical Tests Kolmogorov-Smirnov tests were performed using the built-in MATLAB ® function kstest with the significance level α = 0.01. Sample distributions were compared against exponential distributions set to have the same mean as the sample. V. COMPETING INTERESTS No competing interest is declared. VI. AUTHOR CONTRIBUTIONS STATEMENT A.D.H. and O.A.I. conceived the study, A.D.H. conducted the experiment(s), A.D.H. and O.A.I. analysed the results. A.D.H. and O.A.I. wrote and reviewed the manuscript. Supplemental Information 1. Model Implementation 1.1 Model Structure We model CRISPR-Cas9 substrate selection as a two-branch process in which unbound Cas9 (assumed to be preloaded with a guide RNA) can select between two possible substrates: a fully matched, on-target substrate (R) and a mismatched, off-target substrate (W). The model structure is agnostic to the number and relative positions of mismatches between the Cas9 guide RNA and the target site DNA. The transitions between model states S i and S j , have rate constants defined in terms of transition state theory. Each rate constant is defined in terms of the energy of the starting state ( E i ) and the corresponding reaction barrier : The term ω is a prefactor that ensures that the rate constant has the proper units ( s −1 ), while β = 1 / ( K B T ) sets the energy scale. The on-target and off-target model branches are related through discrimination factors f i = k i,W /k i,R , which fix the ratio between the on-target and off-target rate constants for reaction i . When the discrimination factor f i = 1, it implies that there is no discrimination between on- and off-target substrates for reaction i . 1.2 Model Fitting The model objective function for fitting is defined as where and y i are the predicted and observed data points, respectively. The constant w i is a weighting factor that allows us to scale the relative contribution of each data point to the overall fit. Additionally, we include a set of penalty terms p i that enforce various constraints on the model parameters (such as upper bounds). The model is fitted to the experimental data using particle swarm optimization (PSO) using the MATLAB ® Global Optimization toolbox function particleswarm . The initial conditions for the particle swarm were set based on a previous best fit model in which the model rate constants were obtained by a direct fit, instead of fitting the corresponding state energies ( E i ) and reaction barriers . 1.3 Model Assumptions and Constraints The model is fitted using several assumptions. First, we assume that the R-loop formation and collapse rates ( k 2 , k −2 , k 7 , and k 8 ) have the same discrimination factors for both the WT and HF1-Cas9 variants. In essence, we assume that the protein modifications made by Kleinstiver et al. [ 3 ] do not materially affect the rates of R-loop formation and collapse. Second, we similarly assume that the PAM unbinding rate k −1, a is unchanged across Cas9 variants. Third, we assume that the initial PAM binding rates ( k 1, a and k 9 ) are not only identical between the on- and off-target substrates but that the on-target rates are identical across Cas9 variants. We make a similar assumption for the microscopic cleavage rate, k clv . We enforce “soft” upper bounds for each rate constant to ensure that they take on physically realistic values after fitting. For each parameter k i , we apply a penalty of the form where H ( x ) is the Heaviside step-function and τ i is the upper bound for the given parameter. The constant α is a scaling factor that allows us to vary the magnitude of the penalty; we generally set α = 1000 to ensure that the upper bounds are always satisfied. Finally, we fix the Cas9 HNH domain transitions in the unbound state to the values reported by Dagdas et al. ( k 1, b ≈ 0.29s −1 , k −1, b ≈ 4.2s −1 , [ 1 ]). Since these reactions occur in the unbound state, there are no corresponding discrimination factors to consider. We enforce these rates by again applying a soft penalty term of the form where T i is the target value for the rate constant k i . 1.4 Fitting Substrate Cleavage Rates We use NucleaSeq data published by Jones et al. [ 2 ] to obtain the effective substrate cleavage rates for different Cas9 variant-substrate pairings. The effective cleavage rates are then used to estimate the mean first-passage time (MFPT) for substrate cleavage as In the Jones experimental setup, the effective limit of detection is approximately 6e4 seconds and is limited by the length of the experimental time-course. Any Cas9-substrate pairs that cleave on timescales longer than the experimental timescale cannot be effectively estimated, as there will be very few cleavage events to detect. To account for these substrates, when fitting the model to the cleavage MFPT data we apply a penalty only if the model predicted MFPT is slower than the detection limit, i.e. where D is the limit of detection and H ( x ) is the Heaviside step function. VII. ACKNOWLEDGEMENTS We wish to thank Yavuz Dagdas and Jennifer Doudna for providing us with their smFRET dataset. Additionally, we wish to thank A. Kolomeisky, G. Bao, P. Murphy, Z. Diao, and R. Butcher for useful discussions. We also wish to thank G. Bao and D. Makarov for useful comments on the manuscript. The research was supported by the Welch Foundation (Grant C-1995 to OAI). This work was supported in part by the Big-Data Private-Cloud Research Cyberinfrastructure MRI-award funded by NSF under grant CNS-1338099 and by Rice University’s Center for Research Computing (CRC). Footnotes ↵ * ahecht{at}rice.edu References [1]. ↵ Janice S. Chen , Yavuz S. Dagdas , Benjamin P. Kleinstiver , Moira M. Welch , Alexander A. Sousa , Lucas B. Harrington , Samuel H. Sternberg , J. Keith Joung , Ahmet Yildiz , and Jennifer A. Doudna . Enhanced proofreading governs CRISPR-Cas9 targeting accuracy . Nature , 550 ( 7676 ): 407 – 410 , October 2017 . OpenUrl CrossRef PubMed [2]. ↵ Thomas J. Cradick , Eli J. Fine , Christopher J. Antico , and Gang Bao . CRISPR/Cas9 systems targeting β-globin and CCR5 genes have substantial off-target activity . Nucleic Acids Research , 41 ( 20 ): 9584 – 9592 , November 2013 . OpenUrl CrossRef PubMed Web of Science [3]. ↵ Yavuz S. Dagdas , Janice S. Chen , Samuel H. Sternberg , Jennifer A. Doudna , and Ahmet Yildiz . A conformational checkpoint between DNA binding and cleavage by CRISPR-Cas9 . Science Advances , 3 ( 8 ): eaao0027 , August 2017 . OpenUrl FREE Full Text [4]. ↵ Behrouz Eslami-Mossallam , Misha Klein , Constantijn V. D. Smagt , Koen V. D. Sanden , Stephen K. Jones , John A. Hawkins , Ilya J. Finkelstein , and Martin Depken . A kinetic model predicts SpCas9 activity, improves off-target classification, and reveals the physical basis of targeting fidelity . Nature Communications , 13 ( 1 ): 1367 , March 2022 . OpenUrl CrossRef PubMed [5]. ↵ Yanfang Fu , Jennifer A. Foden , Cyd Khayter , Morgan L. Maeder , Deepak Reyon , J. Keith Joung , and Jeffry D. Sander . High-frequency off-Target mutagenesis induced by CRISPR-Cas nucleases in human cells . Nature Biotechnology , 31 ( 9 ): 822 – 826 , September 2013 . OpenUrl CrossRef PubMed [6]. ↵ Yanfang Fu , Jeffry D. Sander , Deepak Reyon , Vincent M. Cascio , and J. Keith Joung . Improving CRISPR-Cas nuclease specificity using truncated guide RNAs . Nature Biotechnology , 32 ( 3 ): 279 – 284 , March 2014 . OpenUrl CrossRef PubMed [7]. ↵ Daniel T. Gillespie . Exact stochastic simulation of coupled chemical reactions . The Journal of Physical Chemistry , 81 ( 25 ): 2340 – 2361 , December 1977 . OpenUrl CrossRef PubMed Web of Science [8]. ↵ Shanzhong Gong , Helen Hong Yu , Kenneth A. Johnson , and David W. Taylor . DNA Unwinding Is the Primary Determinant of CRISPR-Cas9 Activity . Cell Reports , 22 ( 2 ): 359 – 371 , January 2018 . OpenUrl CrossRef PubMed [9]. ↵ Irina V. Gopich and Attila Szabo . Theory of the statistics of kinetic transitions with application to single-molecule enzyme catalysis . The Journal of Chemical Physics , 124 ( 15 ): 154712 , April 2006 . OpenUrl CrossRef PubMed [10]. ↵ John M. Hinz , Marian F. Laughery , and John J. Wyrick . Nucleosomes Selectively Inhibit Cas9 Off-target Activity at a Site Located at the Nucleosome Edge . The Journal of Biological Chemistry , 291 ( 48 ): 24851 – 24856 , November 2016 . OpenUrl Abstract / FREE Full Text [11]. ↵ Patrick D. Hsu , David A. Scott , Joshua A. Weinstein , F. Ann Ran , Silvana Konermann , Vineeta Agarwala , Yinqing Li , Eli J. Fine , Xuebing Wu , Ophir Shalem , Thomas J. Cradick , Luciano A. Marraffini , Gang Bao , and Feng Zhang . DNA targeting specificity of RNA-guided Cas9 nucleases . Nature Biotechnology , 31 ( 9 ): 827 – 832 , September 2013 . OpenUrl CrossRef PubMed [12]. ↵ Fuguo Jiang and Jennifer A. Doudna . CRISPR-Cas9 Structures and Mechanisms . Annual Review of Biophysics , 46 : 505 – 529 , May 2017 . OpenUrl CrossRef PubMed [13]. ↵ Martin Jinek , Krzysztof Chylinski , Ines Fonfara , Michael Hauer , Jennifer A. Doudna , and Emmanuelle Charpentier . A programmable dual-RNA-guided DNA endonuclease in adaptive bacterial immunity . Science ( New York, N.Y .), 337 ( 6096 ): 816 – 821 , August 2012 . OpenUrl Abstract / FREE Full Text [14]. ↵ Stephen K. Jones , John A. Hawkins , Nicole V. Johnson , Cheulhee Jung , Kuang Hu , James R. Rybarski , Janice S. Chen , Jennifer A. Doudna , William H. Press , and Ilya J. Finkelstein . Massively parallel kinetic profiling of natural and engineered CRISPR nucleases . Nature Biotechnology , 39 ( 1 ): 84 – 93 , January 2021 . OpenUrl CrossRef PubMed [15]. ↵ Nahye Kim , Hui Kwon Kim , Sungtae Lee , Jung Hwa Seo , Jae Woo Choi , Jinman Park , Seonwoo Min , Sungroh Yoon , Sung-Rae Cho , and Hyongbum Henry Kim . Prediction of the sequence-specific cleavage activity of Cas9 variants . Nature Biotechnology , 38 ( 11 ): 1328 – 1336 , November 2020 . OpenUrl CrossRef PubMed [16]. ↵ Misha Klein , Behrouz Eslami-Mossallam , Dylan Gonzalez Arroyo , and Martin Depken . Hybridization Kinetics Explains CRISPR-Cas Off-Targeting Rules . Cell Reports , 22 ( 6 ): 1413 – 1423 , February 2018 . OpenUrl CrossRef PubMed [17]. ↵ Benjamin P. Kleinstiver , Vikram Pattanayak , Michelle S. Prew , Shengdar Q. Tsai , Nhu T. Nguyen , Zongli Zheng , and J. Keith Joung . High-fidelity CRISPR-Cas9 nucleases with no detectable genome-wide off-Target effects . Nature , 529 ( 7587 ): 490 – 495 , January 2016 . OpenUrl CrossRef PubMed [18]. ↵ Mu-Sen Liu , Shanzhong Gong , Helen-Hong Yu , Kyungseok Jung , Kenneth A. Johnson , and David W. Taylor . Engineered CRISPR/Cas9 enzymes improve discrimination by slowing DNA cleavage to allow release of off-Target DNA . Nature Communications , 11 ( 1 ): 3576 , July 2020 . OpenUrl CrossRef PubMed [19]. ↵ Joel D. Mallory , Anatoly B. Kolomeisky , and Oleg A. Igoshin . Trade-Offs between Error, Speed, Noise, and Energy Dissipation in Biological Processes with Proofreading . The Journal of Physical Chemistry. B , 123 ( 22 ): 4718 – 4725 , June 2019 . OpenUrl CrossRef PubMed [20]. ↵ Joel D. Mallory , Anatoly B. Kolomeisky , and Oleg A. Igoshin . Kinetic control of stationary flux ratios for a wide range of biochemical processes . Proceedings of the National Academy of Sciences of the United States of America , 117 ( 16 ): 8884 – 8889 , April 2020 . OpenUrl Abstract / FREE Full Text [21]. ↵ Matthew D. Newton , Benjamin J. Taylor , Rosalie P. C. Driessen , Leonie Roos , Nevena Cvetesic , Shenaz Allyjaun , Boris Lenhard , Maria Emanuela Cuomo , and David S. Rueda . DNA stretching induces Cas9 off-target activity . Nature Structural & Molecular Biology , 26 ( 3 ): 185 – 192 , March 2019 . OpenUrl CrossRef PubMed [22]. ↵ Martin Pacesa , Luuk Loeff , Irma Querques , Lena M. Muckenfuss , Marta Sawicka , and Martin Jinek . R-loop formation and conformational activation mechanisms of Cas9 . Nature , 609 ( 7925 ): 191 – 196 , September 2022 . OpenUrl CrossRef PubMed [23]. ↵ Jeffrey A. Ruffolo , Stephen Nayfach , Joseph Gallagher , Aadyot Bhatnagar , Joel Beazer , Riffat Hussain , Jordan Russ , Jennifer Yip , Emily Hill , Martin Pacesa , Alexander J. Meeske , Peter Cameron , and Ali Madani . Design of highly functional genome editors by modeling the universe of CRISPR-Cas sequences , April 2024 . [24]. ↵ Rohit Satija , Alexander M. Berezhkovskii , and Dmitrii E. Makarov . Broad distributions of transition-path times are fingerprints of multidimensionality of the underlying free energy landscapes . Proceedings of the National Academy of Sciences of the United States of America , 117 ( 44 ): 27116 – 27123 , November 2020 . OpenUrl Abstract / FREE Full Text [25]. ↵ Alexey A. Shvets and Anatoly B. Kolomeisky . Mechanism of Genome Interrogation: How CRISPR RNA-Guided Cas9 Proteins Locate Specific Targets on DNA . Biophysical Journal , 113 ( 7 ): 1416 – 1424 , October 2017 . OpenUrl CrossRef PubMed [26]. ↵ Digvijay Singh , Samuel H. Sternberg , Jingyi Fei , Jennifer A. Doudna , and Taekjip Ha . Real-time observation of DNA recognition and rejection by the RNA-guided endonuclease Cas9 . Nature Communications , 7 : 12778 , September 2016 . OpenUrl CrossRef PubMed [27]. ↵ Digvijay Singh , Yanbo Wang , John Mallon , Olivia Yang , Jingyi Fei , Anustup Poddar , Damon Ceylan , Scott Bailey , and Taekjip Ha . Mechanisms of improved specificity of engineered Cas9s revealed by single-molecule FRET analysis . Nature Structural & Molecular Biology , 25 ( 4 ): 347 – 354 , April 2018 . OpenUrl CrossRef PubMed [28]. ↵ Ian M. Slaymaker , Linyi Gao , Bernd Zetsche , David A. Scott , Winston X. Yan , and Feng Zhang . Rationally engineered Cas9 nucleases with improved specificity . Science ( New York, N.Y .), 351 ( 6268 ): 84 – 88 , January 2016 . OpenUrl Abstract / FREE Full Text [29]. ↵ Christopher A. Vakulskas , Daniel P. Dever , Garrett R. Rettig , Rolf Turk , Ashley M. Jacobi , Michael A. Collingwood , Nicole M. Bode , Matthew S. McNeill , Shuqi Yan , Joab Camarena , Ciaran M. Lee , So Hyun Park , Volker Wiebking , Rasmus O. Bak , Natalia Gomez-Ospina , Mara Pavel-Dinu , Wenchao Sun , Gang Bao , Matthew H. Porteus , and Mark A. Behlke . A high-fidelity Cas9 mutant delivered as a ribonucleoprotein complex enables efficient gene editing in human hematopoietic stem and progenitor cells . Nature Medicine , 24 ( 8 ): 1216 – 1224 , August 2018 . OpenUrl CrossRef PubMed [30]. ↵ Jan-Willem van de Meent , Jonathan E. Bronson , Chris H. Wiggins , and Ruben L. Gonzalez . Empirical Bayes methods enable advanced population-level analyses of single-molecule FRET experiments . Biophysical Journal , 106 ( 6 ): 1327 – 1337 , March 2014 . OpenUrl CrossRef PubMed Web of Science [31]. ↵ Jan-Willem van de Meent , Jonathan E. Bronson , Frank Wood , Ruben L. Gonzalez , and Chris H. Wiggins . Hierarchically-coupled hidden Markov models for learning kinetic rates from single-molecule data . JMLR workshop and conference proceedings , 28 ( 2 ): 361 – 369 , 2013 . OpenUrl [32]. ↵ Lance A. Waller and Bruce W. Turnbull . Probability Plotting with Censored Data . The American Statistician , 46 ( 1 ): 5 – 12 , February 1992 . OpenUrl CrossRef References [1]. Yavuz S. Dagdas , Janice S. Chen , Samuel H. Sternberg , Jennifer A. Doudna , and Ahmet Yildiz . A conformational checkpoint between DNA binding and cleavage by CRISPR-Cas9 . Science Advances , 3 ( 8 ): eaao0027 , August 2017 . OpenUrl FREE Full Text [2]. Stephen K. Jones , John A. Hawkins , Nicole V. Johnson , Cheulhee Jung , Kuang Hu , James R. Rybarski , Janice S. Chen , Jennifer A. Doudna , William H. Press , and Ilya J. Finkelstein . Massively parallel kinetic profiling of natural and engineered CRISPR nucleases . Nature Biotechnology , 39 ( 1 ): 84 – 93 , January 2021 . OpenUrl CrossRef PubMed [3]. Benjamin P. Kleinstiver , Vikram Pattanayak , Michelle S. Prew , Shengdar Q. Tsai , Nhu T. Nguyen , Zongli Zheng , and J. Keith Joung . High-fidelity CRISPR-Cas9 nucleases with no detectable genome-wide off-Target effects . Nature , 529 ( 7587 ): 490 – 495 , January 2016 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted January 29, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Kinetic mechanism for fidelity of CRISPR-Cas9 variants Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Kinetic mechanism for fidelity of CRISPR-Cas9 variants Andrew D. Hecht , Oleg A. Igoshin bioRxiv 2025.01.27.634389; doi: https://doi.org/10.1101/2025.01.27.634389 Share This Article: Copy Citation Tools Kinetic mechanism for fidelity of CRISPR-Cas9 variants Andrew D. Hecht , Oleg A. Igoshin bioRxiv 2025.01.27.634389; doi: https://doi.org/10.1101/2025.01.27.634389 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Biophysics Subject Areas All Articles Animal Behavior and Cognition (7617) Biochemistry (17633) Bioengineering (13856) Bioinformatics (41840) Biophysics (21398) Cancer Biology (18529) Cell Biology (25422) Clinical Trials (138) Developmental Biology (13352) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24281) Genetics (15582) Genomics (22461) Immunology (17700) Microbiology (40289) Molecular Biology (17138) Neuroscience (88413) Paleontology (666) Pathology (2823) Pharmacology and Toxicology (4813) Physiology (7632) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4283) Systems Biology (9807) Zoology (2267)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.