Full text
34,613 characters
· extracted from
preprint-html
· click to expand
Identifying Factors Important for Conservation at Sites of Synonymous Variations | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Identifying Factors Important for Conservation at Sites of Synonymous Variations Abhirami Ramasubramanian , Uma Sunderam , Rajgopal Srinivasan doi: https://doi.org/10.1101/2024.01.01.573819 Abhirami Ramasubramanian 1 TCS Research and Innovation , Deccan Park, Madhapur, Hyderabad-500081, INDIA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Uma Sunderam 1 TCS Research and Innovation , Deccan Park, Madhapur, Hyderabad-500081, INDIA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Rajgopal Srinivasan 1 TCS Research and Innovation , Deccan Park, Madhapur, Hyderabad-500081, INDIA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: rajgopal.srinivasan{at}tcs.com Abstract Full Text Info/History Metrics Preview PDF Abstract Synonymous mutations can have a deleterious effect leading to disease, even though they are not protein altering. Variations at genomic sites leading to synonymous variants are frequently highly conserved across species. Several prediction methods have been developed to assess the impact of synonymous mutations and are highly dependent on having validated sets of both deleterious and benign synonymous mutations. However, validated data available for deleterious synonymous mutations is sparse unlike for missense mutations. Rather than develop a model for predicting pathogenicity of synonymous variants, we seek to understand the relative importance of various factors that lead to conservation at sites of synonymous variants. Our study built machine learning models using various features on a large set of reported and generated synonymous variants ( Zeng Z et al, 2019 ) to predict conservation (Genomic Evolutionary Rate Profiling – Rejected Substitution (GERP RS) base scores and Phylogenetic p-values for 100 vertebrates (PP100)) at genomic sites. We used the extreme gradient boosting classifier to classify sites as high, medium and low conservation at different cutoffs. Our experiments report an AUC between 0.74-0.79 and the sensitivity was significant. Of the features we explored, a few alternate allele independent properties were repeatedly flagged as having high impact. These findings provide information for predictors to further improve models for synonymous variant impact. Introduction Variants causing a direct change in the transcribed protein such as missense, nonsense, frame-shift insertion deletions, and splice site disrupting constitute the majority of pathogenic variants in many disease databases. Synonymous mutations can also have a deleterious effect on function even though they are not protein altering. Variations at synonymous sites can impact splicing including cryptic splice, splice enhancers and suppressors, disrupt transcription, co-translational folding and mRNA stability among other things. Several machine learned models have been developed to assess the impact of synonymous mutations and are highly dependent on having validated sets of deleterious and benign synonymous mutations ( Zeng Z et al, 2019 ; Buske OJ et al, 2013 ; Livingstone M et al, 2017 ; Livingstone M et al, 2017 ; Shi F et al, 2019 ). However, validated data available for deleterious synonymous mutations is sparse unlike for missense mutations. For e.g. ClinVar database (accessed 30 th April, 2022) reports only 318 pathogenic synonymous variants in comparison to 56,100 pathogenic non-synonymous variants (missense and nonsense). Rather than develop a model for predicting pathogenicity of synonymous variants, we seek to understand the relative importance of various factors that lead to conservation at sites of synonymous variants. Our study built machine learning models using a variety of features on a large set of reported and generated synonymous variants that were used in another study ( Zeng Z et al, 2019 ) to predict the conservation score at the variant site. As evidenced ( fig 1a ) by analyzing synonymous ClinVar mutations, GERP RS (Rejected Substitution) base scores ( Cooper GM et al, 2005 ; Davydov EV et al, 2010 ) and PhyloP values for 100 vertebrates (PP100) ( Pollard, K.S. et al, 2009 ) provide good, albeit imperfect, discrimination between benign and pathogenic variants. We therefore postulate that factors important for predicting conservation may also play an important role in predicting functional deleteriousness. We report the results of our study. Download figure Open in new tab Figure 1: a) GERP RS distributions of benign and pathogenic ClinVar synonymous variants b) GERP RS distributions of synonymous and missense ClinVar variants c) PP100 distributions of benign and pathogenic ClinVar synonymous variants d) PP100 distributions of synonymous and missense ClinVar variants Materials and Methods Datasets (i) Synonymous Variants data Observed and generated synonymous variants that were used in a previous study ( Zeng Z et al, 2019 ) were combined to form the data set for the study. The observed variants (1,362,607 synonymous variants) were collected from public databases such as 1000 Genomes (The 1000 Genomes Project Consortium, 2015), ExAC ( Lek M et al, 2016 ) and gnomAD ( Karczewski, K.J. et al 2020 ). The generated variants were a size matched random subset of all possible synonymous variants. Duplicates in the combined data were excluded. (ii) Variant Selection A genomic variant site can have a different mutation type depending on the transcript. We selected our synonymous variants from the above set considering only those synonymous variants that occured in Matched Annotation from NCBI and EMBL-EBI (MANE) transcripts ( Morales J. et al, 2022 ). The annotations were from GENCODE ( Frankish A et al, 2021 ) project (version 39 lifted over to GRCh37) and mutation effect as annotated from our in-house script Varant ( http://compbio.berkeley.edu/proj/varant/Home.html ). In order to restrict the variant effect to purely synonymous sites, we only retained those where all three possible allele changes were synonymous in the thus prioritized transcript. Variant sites that were within 3 bases of splice acceptor or donor sites were excluded from our experiments to avoid any bias caused by direct disruption of the splice region. Moreover, only variants that had annotations for all the features were retained. For e.g. if by virtue of being in the first or last exon, a variant did not have a splice acceptor or donor score for the exon, it was removed from the data set. The final dataset consisted of 7,24,568 sites of synonymous variation. Features A total of 35 features were used and briefly described below. (i) Position based Features that describe the location of the site include: 1) the position of the site in the exon, 2) the position of the corresponding amino acid in the protein, and 3) whether the site falls in functional regions such as transcription factor binding sites. (ii) Splice based Many variants have deleterious effects because of their impact on splicing. Some of the splicing-related features we have considered include: 1) the MaxEntScan ( Yeo G et al, 2004 ; Eng L et al, 2004 ) donor and acceptor score percentiles of the splice sites and 2) whether or not cryptic splice sites were created by any of the alternate alleles at that position and if they were, their respective MaxEntScan scores. The MaxEntScan percentiles were calculated by computing the MaxEntScan scores of all unique splice acceptors and donors from Gencode transcripts. We also considered the impact of the variants on Exonic Splice Enhancers (ESE) and Exonic Splice Silencers (ESS) ( Fairbrother WG et al, 2002 ; Wang, Z et al, 2004 ). A count of the number of ESEs lost and ESSs gained and the ESR scores (average hexamer ESRseq scores ( Ke S et al, 2011 ) computed over a sliding window over the site) for the reference and alternate alleles as well as their difference were the features used to study this. (iii) Codon based The codon based features consist of codon usage features such as the Relative Synonymous Codon Usage (RSCU) calculated for the reference codon and all alternate codons, along with their respective differences. RSCU is defined as the ratio of the observed frequency of codons to the expected frequency given that all the synonymous codons for the same amino acids are used equally ( Eq 1 ) ( Nakamura Y et al, 2000 ). We have calculated RSCU globally as well as specific to the transcript considered for each site. where Xi is the number of occurrences and RSCUi is the relative synonymous codon usage for codon i. Codon related features that could potentially impact translation were also added to the set. Specifically, the additional occurence of the reference codon before or after and contiguous occurence with respect to the variant codon site. To assess the impact of the actual codon, the reference codon was also encoded base-wise in terms of the purine/pyrimidine bases and double or triple hydrogen bondedness. All the features are described in detail in Supplementary Table 1 . View this table: View inline View popup Download powerpoint Table 1: Average metrics from 10-fold cross validations and 1 run of the independent test sets for the 4 different models. Alternate allele features Each sample represented a site rather than a variant since the conservation score was for the site and not alternate allele dependent. Of the alternate allele specific features, the variant which had the hypothesised maximum impact, for e.g., the maximum ESS gain among 3 alternate alleles was selected for each site. Similarly, minimum of the ESR and RSCU alternate allele scores were considered amongst the variants. Variants creating cryptic donor sites were considered as having higher impact over those creating cryptic acceptor sites. Within cryptic splice sites of the same type, the allele that resulted in the maximum maxent cryptic donor or acceptor score percentile was selected. Model Building Models were built using two different measures of conservation as the target variable. In one, the base-wise GERP RS conservation scores were used as the target variable and in another, PhyloP values for 100 vertebrates ( Pollard, K.S. et al, 2009 ). Based on the observed distribution ( Figures 1b and 1d ) of reported synonymous variants in clinvar database, the data was divided into 3 classes of variants based on their evolutionary conservation scores. These corresponded to the high class (highly conserved), low class (low conservation) and medium class (variants with conservation scores between the high and low classes). Two different distributions of the data were used: Case 1) high class having conservation scores above the 90th percentile and the low class with conservations scores in the bottom 10 th percentile and Case 2) high class scores above the 3 rd quartile and low class conservation scores below the 1 st quartile (cutoffs in Supplementary Table 2 ). The remaining variants in between the high and low classes were considered the medium class in both cases. These precomputed scores were obtained from http://mendel.stanford.edu/SidowLab/ downloads/gerp/hg19.GERP_scores.tar.gz in the case of GERP RS and https://hgdownload.soe.ucsc.edu/goldenPath/hg19/phyloP100way/hg19.100way.phyloP100way.bw in case of PhyloP100 way. Regression models were also built and compared (results not shown) but were discarded in favour of the multiclass classifier because of inadequate prediction accuracy. View this table: View inline View popup Download powerpoint Table 2: Significant features reported by SHAP and feature importances The XGBoost package ( Chen T et al, 2016 ) version 1.7.3, a gradient boosting library designed to be a highly efficient implementation, was used to train and test the data. 3 class models based on the above data were built with number of estimators set to 1000 and other default parameter values. Results Model Evaluation The model was evaluated with both 10-fold stratified cross validation as well as an independent test set (a randomly selected 20% of the total dataset). The model distinguished highly conserved and non-conserved variants with an AUC of between 0.74 and 0.79 ( Table 1 ). Similar AUC and balanced accuracy values were observed for the test set ( Table 1 ). Significant Features We next looked into significant features and their relative importances for the 4 models with SHapley Additive exPlanations (SHAP) values ( Lundberg, S.M. et al, 2020 ), an approach to explain the output of machine learning models based on classical Shapley values from cooperative game theory and their related extensions. For each model, the top 15 features across the 3 classes were considered and a union across the 4 models resulted in 17 significant features. The top 15 features across the 3 classes in one of the models is shown in Figure 2 . In addition, the top 15 features from feature importances module in scikit learn were also compared from the 10-fold cross validation data and 3 additional exclusive features were added to the list. The significant features are consolidated in Table 2 . Download figure Open in new tab Fig 2: Shap Value plots for PP100 target model with >= 90 percentile high class. Important features based on mean SHAP values across the 3 different classes (box plot) and individual classes (beeswarm plots). clockwise from top left: a) box plot for significant features b) beeswarm plot for high class c) beeswarm plot for low class and d) beeswarm plot of middle class The features that were consistently reported as significant were related to RSCU scores (both global and transcript based), ESR scores, base identity at substitution site, TFBS, exon length and codon count. RSCU (global value) for the reference allele was the top feature in 3 out of 4 models as per feature importances. It was also observed that for other feature groups like RSCU (transcript based) and ESR score, the reference feature frequently ranked highest among the set, ahead of the difference with the alternate allele and alternate allele values. The identity of the 2 nd and 3 rd reference codon base were also consistently in the top 15 features. Other features that were reported as significant albeit lower ranked in the top 15 were TFBS, exon length and codon count. The models were also built based on annotations using GRCh38 reference databases and similar trends were observed ( Supplementary Table 3 ). View this table: View inline View popup Download powerpoint Table 3: Significant related feature groups performance on the independent test set We next examined how groups of related features and individual features performed for a model built with PP100 as target and high class >= 90 percentile and low class < 10 percentile. We also built models with all the 20 short listed features and with only the top 10 best performing features based on the individual feature models. The results of the test run are reported in Table 3 . Of the 3 related feature groups, Codon usage performed the best with an AUC of almost 0.75. A few individual features – rscu_ref and rscu_diff_min marginally outperformed both the reference codon related and splice related groups ( Supplementary Table 4 ). Reducing the feature set to only the top 10 features based on the single feature models (not shown), gave a performance almost on par with the full feature set. With all 3 related feature groups together (17 features), the performance difference with using all 35 features was negligible. View this table: View inline View popup Download powerpoint Table 4: Test set results for Models built for single exon, first exon and last exons Region-wise models In order to minimize features without data, our model had excluded variants located in the first exon, last exon, and variants located in single exon genes. In order to include such variants and verify the model capability of distinguishing high, medium and low conservation, we built specific models for such regions with only applicable features. Models for all 3 special cases of variants; viz; occuring in the first, last and single exon genes could distinguish between the high, medium and low conservation ( Table 4 ). The model for first exon performed slightly below par with an AUC of between 0.69 to 0.74 compared to the other models. The last exon model could pick higher conservation variants better compared to all other models (precision). Next, top 10 significant features that impacted the prediction were evaluated using SHAP as before. Although the overlap with features identified by the general models were similar, the order of importance of few individual features were different ( Supplementary Figure 1 ). For e.g length of exon rose in significance especially for the single exon models and position in the exon also had a higher weightage. Significant Features with Other Predictors In our experiments thus far, we had consistently observed some features as significant. These included Codon usage related, ESR scores and actual nucleotide at the third position of the codon. To assess the impact of a subset of features on available predictor algorithms, we experimented with adding combinations of our features to other features used by the predictors and also replacing the conservation component with the features to evaluate the performance. The Identification of Deleterious Synonymous Variants (IDSV) model (Shi F, et al 2019) uses the random forest method to identify deleterious synonymous variants with 10 optimized features. We have attempted replicating their results with their data and code and further annotated their datasets with significant features identified in this study. We test whether top features can replace their conservation feature, PhyloP, and still retain performance. Since our main model excluded near splice variants, the variants selected for the training and test sets from the IDSV data excluded such variants. We retained synonymous SNVs from the IDSV train and test sets which occured in MANE transcripts. This resulted in a total of 418 variants in the training set (down from their original training set of 600), with 163 deleterious and 255 benign variants. Since we couldn’t replicate their Translation Efficiency feature values, we calculated these values using the MANE transcript that was selected for the variant. All other features were annotated and models were run as specified in their README and using the downloaded scripts. This experiment was carried out with the variant information and not the genomic sites as with our models since IDSV is a variant effect predictor. Our test data for the experiment comprised of the IDSV test set, augmented with an independent set of pathogenic and benign synonymous ClinVar variants (accessed 30th April 2022). The synonymous clinvar variants were shortlisted as follows: ‘Pathogenic’ & ‘Pathogenic/Likely_pathogenic’ sSNVs for the deleterious class and ‘Benign’ sSNVs for the benign class were selected from the file, without checking for assertion. The ClinVar variants were filtered in the same way as the IDSV train and test sets as described previously and then annotated with the 10 IDSV features as well as our top 19 features (excluding TFBS as IDSV uses it as one of its 10 optimized features). The unique set of combined IDSV test and ClinVar variants not found in the train set was the test set used for our experiment. The combined test set had 163 deleterious and 39,448 benign sSNVs. We observed that all metrics reported were lower when PhyloP was dropped as a feature ( Table 5 ). However when the individual features shortlisted in our model were added, for most of the cases, the AUC increased ( Supplementary Table 5 ). The performance after adding the 3 related features groups individually was comparable to the original dataset with PhyloP. Next, the top 10 best performing features, 3 related feature groups combined, and all shortlisted 20 features were added and the prediction run for the test set. In these cases, the performance exceeded that of the original model with all 10 IDSV features (including PhyloP). View this table: View inline View popup Download powerpoint Table 5: IDSV performance (independent test set combining IDSV test and clinvar variants) with different combinations of feature groups identified as significant in place of conservation score PhyloP 100 way conservation score feature (average values over 10 runs) Discussion We evaluated conservation classification as a proxy for variant deleterious effect using 35 features categorized into codon related, potential splicing disrupting and relative position within the protein. A large dataset with variants that were a mixture of reported and possible synonymous mutations were used for the multi class (high, medium and low conservation) classification and showed AUC, precision, recall and F1-score values that were higher than with random classification. Both 10-fold stratified cross validation and evaluation with an independent test set gave similar results. When importances of the features in the model were examined, the top features were consistent across cutoffs and over multiple runs of the cross validation. It was notable that some significant features were reference based and ranked as important along with the alternate allele value and the difference between the reference and alternate allele of the related feature, the most striking being RSCU. Codon usage is considered among the important features and used as part of the feature set in synonymous effect prediction ( Buske OJ et al, 2013 ; Livingstone M et al 2017; Zhang X et al 2017). It was however, observed that the RSCU of alternate allele and the difference with the reference is used for the prediction and the reference value is not used by any predictor. In addition to global RSCU values, transcript based RSCU values were also found to be significant. In addition to splice regulation disruption annotation that is used in synonymous effect prediction, we used the ESR score of the reference, alternate allele and their difference. A similar pattern was observed for this feature group with ESR score of the reference allele being reported as important in addition to the ESR score of the alternate and the difference between the reference and alternate. The identity of the 2 nd and 3 rd bases in the reference codon was significant in several of the models. We speculate that these could have direct correlation to the conservation of that position. Apart from the above features, the length and distance of the variant from the exon end also contributed to the model prediction. Our results suggest that there can be inherent significance in the reference that is independent of the alternate allele. Feature selection for some genomic properties might thus benefit from such expansion of the underlying features and the inclusion of reference features in model building. We started with a basic set of commonly used features and the approach can be expanded to other features for evaluation. A class segregation that more closely mimics reality would be an additional transitional class with moderate conservation in addition to high conservation and low conservation classes. Our models built around this performed almost at par and features underlying the model remained similar. Variant predictors could also extend the models to cover variants in the first and last exons and single exon genes which might potentially show different behaviour based on their location and require separate models with a corresponding set of features. Indeed, other predictors (Livingstone M et al, 2017), have observed differences in the significant features and prediction performance based on the distance of variants along the exon and proposed different models to address this. Features selected from our model showed a reasonable replacement for conservation in a synonymous effect prediction tool and showed potential for further improvement of prediction when added to the orginal features of the tool. Conclusion Our novel approach helped identify key features that can be used to build better models for synonymous mutations. Moreover, the identification of significant reference features that are independent of alternate alleles provides interesting avenues for further exploration to identify vulnerable sites in the genome and also gain insight into possible disease mechanisms. Supplementary Tables and Figures Download figure Open in new tab Supplementary Fig 1: Significant features based on mean SHAP values for (clockwise) single exon, first exon, last exon and general models with high class PP100 >= 90 percentile View this table: View inline View popup Supplementary Table 1: Features used to build the 3 class models View this table: View inline View popup Download powerpoint Supplementary Table 2: Class cutoff values for GERP and PhyloP models View this table: View inline View popup Download powerpoint Supplementary Table 3: GRCh37 vs GRCh38 results from independent test set View this table: View inline View popup Supplementary Table 4: Individual feature models performance with high class PhyloP100 >= 90 percentile of dataset. View this table: View inline View popup Supplementary Table 5: Test set results with IDSV features without conservation (PhyloP) feature with individual features and feature groups added (average reults over 10 runs) References 1. ↵ Zeng Z and Bromberg Y . Predicting Functional Effects of Synonymous Variants: A Systematic Review and Perspectives . Front. genet . 2019 ; 10 (914) 2. ↵ Buske O J , Manickaraj A et al. Identification of deleterious synonymous variants in human genomes . Bioinf . 2013 ; 29 (15) 3. ↵ Livingstone M , Folkman L et al. Investigating DNA, RNA and protein-based features as a means to discriminate pathogenic synonymous variants . Hum Mutat . 2017 ; 38 (10) 4. ↵ Zhang X , Li M et al. regSNPs-splicing: a tool for prioritizing synonymous single-nucleotide substitution . Hum Genet . 2017 ; 136 5. ↵ Shi F , Yao Y et al. Computational identification of deleterious synonymous variants in human genomes using a feature-based approach. BMC Medical Genomics . 2019 , 12 6. Landrum MJ , Lee JM et al. ClinVar: improving access to variant interpretations and supporting evidence . Nucleic Acids Res . 2018 Jan 4. ( clinvar ) 7. ↵ Cooper GM , Stone EA et al. Distribution and intensity of constraint in mammalian genomic sequence . Genome Res . 2005 ; 15 : 901 – 913 ( gerp ) OpenUrl Abstract / FREE Full Text 8. ↵ Davydov EV , Goode DL et al. Identifying a High Fraction of the Human Genome to be under Selective Constraint Using GERP++. PLoS Comput Biol . 2010 ; 6 ( gerp ) 9. A global reference for human genetic variation , The 1000 Genomes Project Consortium , Nature 526 , 68-74 (01 October 2015 ). OpenUrl 10. ↵ Lek M. , Karczewski K.J . et al. Analysis of protein-coding genetic variation in 60, 706 humans . Nature . 2016 ; 536 : 285–291 . ( exac ) OpenUrl 11. ↵ Karczewski , K.J. , Francioli , L.C. , Tiao , G. , et al. The mutational constraint spectrum quantified from variation in 141,456 humans. Nature 581 , 434–443 ( 2020 ) ( gnomad ) OpenUrl 12. ↵ Morales , J. , Pujar , S ., et al. A joint NCBI and EMBL-EBI transcript set for clinical genomics and research. Nature 604 , 310 – 315 ( 2022 ) OpenUrl 13. ↵ Frankish A , Diekhans M et al. GENCODE 2021 . Nucleic Acids Res 2021 : 49 ; d1; D916-D923 OpenUrl 14. ↵ Yeo G , Burge C B. Maximum entropy modeling of short sequence motifs with applications to RNA splicing signals . J Comput Biol . 2004 ; 11 ( 2-3 ) ( MAXENT ) 15. ↵ Eng L , Coutinho G et al. Nonclassical splicing mutations in the coding and noncoding regions of the ATM Gene: maximum entropy estimates of splice junction strengths . Hum Mutat . 2004 Jan ; 23 ( 1 ) ( MAXENT ) 16. ↵ Fairbrother W G , Yeh R F , et al. Predictive identification of exonic splicing enhancers in human genes. Science . 2002 ; 297 ( 5583 ) ( ESE ) 17. ↵ Wang , Z , Rolish , et al. Systematic identification and analysis of exonic splicing silencers . Cell . 2004 ; 119 ( ESS ) 18. ↵ Ke S , Shang S et al. Quantitative evaluation of all hexamers as exonic splicing elements . Genome res . 2011 ; 21 ( 8 ) ( ESR ) 19. ↵ Nakamura Y , Gojobori T , Ikemura T. Codon usage tabulated from international DNA sequence databases: status for the year 2000 . Nucleic Acids Res . 2000 ; 28 ( 1 ) 20. ↵ Pollard , K. S. , Hubisz , M. J. et al. ( 2009 ). Detection of nonneutral substitution rates on mammalian phylogenies . Genome Res . 20 , 110 – 121 . ( phylop ) OpenUrl PubMed Web of Science 21. ↵ Chen T and Guestrin C. XGBoost: A Scalable Tree Boosting System . KDD ‘16: Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining . August 2016 Pages 785 – 794 . 22. ↵ Lundberg , S.M. , Erion , G. , Chen , H ., et al. From local explanations to global understanding with explainable AI for trees. Nat Mach Intell 2 , 56 – 67 ( 2020 ). OpenUrl View the discussion thread. Back to top Previous Next Posted January 02, 2024. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Identifying Factors Important for Conservation at Sites of Synonymous Variations Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Identifying Factors Important for Conservation at Sites of Synonymous Variations Abhirami Ramasubramanian , Uma Sunderam , Rajgopal Srinivasan bioRxiv 2024.01.01.573819; doi: https://doi.org/10.1101/2024.01.01.573819 Share This Article: Copy Citation Tools Identifying Factors Important for Conservation at Sites of Synonymous Variations Abhirami Ramasubramanian , Uma Sunderam , Rajgopal Srinivasan bioRxiv 2024.01.01.573819; doi: https://doi.org/10.1101/2024.01.01.573819 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7647) Biochemistry (17729) Bioengineering (13922) Bioinformatics (42050) Biophysics (21490) Cancer Biology (18637) Cell Biology (25569) Clinical Trials (138) Developmental Biology (13404) Ecology (19943) Epidemiology (2067) Evolutionary Biology (24368) Genetics (15625) Genomics (22550) Immunology (17764) Microbiology (40476) Molecular Biology (17208) Neuroscience (88766) Paleontology (667) Pathology (2843) Pharmacology and Toxicology (4834) Physiology (7660) Plant Biology (15175) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9836) Zoology (2272)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.