Full text
30,392 characters
· extracted from
preprint-html
· click to expand
NucleoSeeker: Precision Filtering of RNA Databases to Curate High-Quality Datasets | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results NucleoSeeker: Precision Filtering of RNA Databases to Curate High-Quality Datasets View ORCID Profile Utkarsh Upadhyay , Fabrizio Pucci , Julian Herold , Alexander Schug doi: https://doi.org/10.1101/2024.12.06.626307 Utkarsh Upadhyay 1 John von Neumann Institute for Computing, Jülich Supercomputing Centre , Jülich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Utkarsh Upadhyay Fabrizio Pucci 2 Computational Biology and Bioinformatics, Université Libre de Bruxelles , Brussels, Belgium 3 Interuniversity Institute of Bioinformatics , Brussels, Belgium Find this author on Google Scholar Find this author on PubMed Search for this author on this site Julian Herold 4 Scientific Computing Center, Karlsruhe Institute for Technology , Karlsruhe, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Alexander Schug 1 John von Neumann Institute for Computing, Jülich Supercomputing Centre , Jülich, Germany 5 Department of Biology, University of Duisburg-Essen , Essen, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: al.schug{at}fz-juelich.de Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract The structural prediction of biomolecules via computational methods complements the often involved wet-lab experiments. Un-like protein structure prediction, RNA structure prediction remains a significant challenge in bioinformatics, primarily due to the scarcity of annotated RNA structure data and its varying quality. Many methods have used this limited data to train deep learning models but redundancy, data leakage and bad data quality hampers their performance. In this work, we present NucleoSeeker, a tool designed to curate high-quality, tailored datasets from the Protein Data Bank (PDB) database. It is a unified framework that combines multiple tools and streamlines an otherwise complicated process of data curation. It offers multiple filters at structure, sequence and annotation levels, giving researchers full control over data curation. Further, we present several use cases. In particular, we demonstrate how NucleoSeeker allows the creation of a non-redundant RNA structure dataset to assess AlphaFold3’s performance for RNA structure prediction. This demonstrates NucleoSeeker’s effectiveness in curating valuable non-redundant tailored datasets to both train novel and judge existing methods. NucleoSeeker is very easy to use, highly flexible and can significantly increase the quality of RNA structure datasets. 1 Introduction Deep learning (DL) technology has given a significant boost to scientific research by providing powerful tools for data analysis, pattern recognition and prediction [ 1 ]. It strongly impacted the computational structural biology community enabling the recent breakthroughs such as AlphaFold [ 2 ] providing a massive improvement in both speed and accuracy for protein structure prediction. Prior methods based on statistical inference such as direct coupling analysis (DCA) [ 3 ] gave a glimpse at the value hidden within the evolution of biomolecular sequences, enabling the statistical inference of spatial adjacencies to guide structure prediction tools [ 4 , 5 ]. Transformer networks as used by models such as AlphaFold [ 2 ] leverage the information of protein evolution as found in sequence data to derive structural information. While these approaches are highly successful [ 6 ], they require abundant training data. Thus, scarcity of data prohibits the direct transfer of these methods to other biomolecules, e.g. RNA. Moreover, the available RNA structural data suffers not only from its limited size but also from high redundancy and low data quality. Specifically, the current version of the Protein Data Bank (PDB) [ 7 ] contains a large number of highly similar RNA structures, structures with poor resolution, a significant number of hybrids (Protein/RNA, DNA/RNA and others) and a considerable proportion of very short sequences (fewer than twenty residues) (see Fig. 1a and Figure 1 in Supplementary Information). Download figure Open in new tab Figure 1. (a, top) Hierarchical classification of RNA data : from 30 million sequences to 121 unique structures. This nested diagram illustrates the progressive filtering of RNA information, showcasing the rarity of well-characterized, unique structures among the vast sea of known sequences. Each layer represents increasingly stringent criteria. (b, middle) Barnacle and PyDCA contact prediction performance : Top-L precision of two RNA contact prediction tools, Barnacle (red circles) and PyDCA (blue triangles), for the 𝒟 𝒞 structures as a function of their Infernal Bit Score with the corresponding RFAM families (x-axis). (c, bottom) AlphaFold pTM score vs RMSD : AlphaFold pTM score as a function of the Root Mean Square Deviation (RMSD) between AlphaFold3 predictions and experimental structures in Å, categorized by sequence identity (SI%) levels. Such properties often cause DL models to overfit and generalize poorly. Assessing DL models is also challenging because data leakage, stemming from improper splits between training and test sets, can lead to an overestimation of the model’s performance [ 8 , 9 ]. Furthermore, reproducibility also becomes impaired. Experiments like RNA-Puzzles [ 10 ] and CASP [ 11 ], where computational algorithms have to blindly predict RNA structures, provide a valid evaluation method. However, results [ 11 ] suggest that due to the above-mentioned problems, DL models currently perform worse than physics-based approaches in RNA structure prediction task. The development of curated datasets for training and testing models is essential to addressing these issues. For instance, following strict filtering processes and manual curation as in [ 12 , 13 ] can be highly effective. Here, we introduce NucleoSeeker, an easy-to-use software that provides extensive flexibility and control for curating RNA datasets from structures deposited in the PDB database in a fully automated manner. 2 Methods NucleoSekeer is a python library that can be directly used as a command-line tool with limited dependencies (i.e. Biopython, Pandas, Numpy and Requests). It handles downloading and applying filters to create a dataset. 2.1 Dataset Access Initially, no structures are downloaded from the RCSB PDB database. Instead, we first use the Search API of the PDB to retrieve all IDs for a specified structure determination method and a given polymer entry type. These IDs are then processed through a GraphQL query, which fetches predefined attributes, such as the experimental method used, resolution and many more for each structure (see section 1 of Supplementary Information for details). This API-based approach ensures that our tool generates the most up-to-date dataset without requiring any code modifications. This module yields a data frame DF with all requested IDs and their corresponding attributes. To refine the dataset, we use three different kinds of filters in our software that allow the users to specify their requirements for various levels from the individual chain to multiple structures; all the filter modules in the package are also available as standalone modules. 2.2 Dataset Creator The results of the filtering operations are combined to generate a dataset of RNA structure. Infernal [ 14 ] is then used to search for RFAM families [ 15 ] of the filtered structures. Users can specify an E-value threshold for the RNA family hits that specifies the statistical significance of the result (refer to the user guide of Infernal [ 14 ] for more details). A lower E-value indicates a more significant result, effectively controlling the strictness of the family search. The output of NucleoSeeker consists of a list of RNA chains along with corresponding information, such as RFAM classification, PDB code and the corresponding number of chains, structure resolution, experimental method used, and year of release. Note that, in the case of complexes, only the RNA chains that meet all the specified criteria are selected. 2.3 Filtering Mechanisms In our filtering approach, we follow the hierarchy used by the RCSB PDB database to organize structures. More in detail, we use the three common levels ENTRY, ENTITY and INSTANCE , which form the basis of the different modules. Here, we will briefly describe the functionality of the different modules and how they integrate to generate the final dataset. An exhaustive list of arguments and parameters for each module can be found in the Supplementary Information. Note, that the word Polymer is used in multiple parameters across modules, and each of them carries a different meaning (see Supplementary Information). Metadata We use this module to apply filters based on the metadata stored in the DF . This offers a comprehensive set of filters for structural attributes, e.g. users can filter structures based on the experimental methods used for determination, such as X-ray diffraction, NMR, or Cryo-EM, ensuring that only structures determined by these techniques are included. The module also offers a resolution threshold filter, allowing users to exclude structures with resolutions outside a specified value. Similarly, structures can be filtered by their release year, allowing the inclusion of only those resolved before a specific year or within a particular range of years. Furthermore, the module supports selection based on polymer entity types, enabling users to customize their datasets by focusing on specific polymers or excluding polymer complexes. Additionally, the keyword filter provides the ability to include or exclude structures based on specific terms, offering further refinement of the dataset to meet users’ requirements. Individual Structure The IndividualStructure module analyzes the composition of structures for further processing, i.e. it parses individual files. It examines each structure’s polymer type, nucleotide or residue integrity, and the length of each chain. This module can function independently to download and verify PDB files based on specified criteria when given a PDB ID. Utilizing a PDB parser, it extracts and analyzes structural information, allowing users to specify parameters such as the desired polymer instance type and sequence length to filter out chains that do not meet these criteria. This module enables the easy removal of short sequences and the extraction of chains containing only RNA. Structure Comparison This filter level combines our filtered DF from the Metadata filter and the IndividualStructure filter by applying the latter to each structure in the DF . We also integrate sequence alignment tools, such as Clustal Omega [ 16 ] and Emboss [ 17 ], to filter structures based on sequence identity (SI). If the SI between two structures exceeds a certain threshold, We select the structure with the highest resolution to minimize redundancy in the dataset while ensuring good overall structure resolution. A sequence identity matrix is created using the specified alignment tools. However, since EMBOSS processes sequences in pairs, impacting performance, it is generally recommended to use Clustal Omega for most applications. These filters, whether used individually or collectively, provide an unprecedented level of control and flexibility in curating datasets from the PDB database. Consequently, researchers can create more targeted and refined datasets, significantly enhancing the accuracy and reliability of their RNA structure prediction protocols. 3 Results Well-curated and non-redundant datasets serve two important goals: they improve training outcomes and are crucial for assessing method performance. Here, we show that NucleoSeeker can be used to attain this goal. In particular, we highlight two examples where our tool can create datasets for assessing RNA contact prediction methods and evaluating AlphaFold3 [ 18 ] performance on RNA structure prediction. The manual curation of such datasets is time-consuming and error-prone and can lead to non-systematic biases. We show the ease of curating such datasets using NucleoSeeker and believe it can be used to prepare datasets for machine learning algorithms in a similar way. 3.1 Use-case 1: Automated RNA structures curation for assessing contact prediction We used NucleoSeeker to create a well-curated and non-redundant dataset starting from all the 7704 RNA structures available in the PDB database (accessed in July 2024). We selected only RNA structures resolved by X-ray crystallography, with a resolution below 3.6 Å and a maximum pair-wise sequence identity of 50%. Since we wanted to create a dataset of RNA-only structures we used ‘pdbx keywords’=‘RNA’ . We ended up with 117 structures, out of which 88 have an associated RFAM family [ 15 ]. These parameters are the same as those used in the construction of the dataset curated in [ 12 ], in which only 69 families were included (PDB database accessed in 2020). We note that although there has been an increase in the number of resolved RNA structures over the years, the data remains scarce and significant improvements are needed to train DL models on these limited data. This dataset labelled with 𝒟 𝒞 is then used to assess the performance of two unsupervised RNA contact prediction methods, namely PyDCA [ 19 ] and Barnacle [ 20 ]. In Fig. 1.b , we present the performance of these two methods on 𝒟 𝒞 , as measured by the Precision at rank L, which represents the proportion of correctly predicted nucleotide contacts among the top L predictions. We note that the two methods reach good performances with Precision equal to about 0.62 for Barnacle and 0.45 for PyDCA when averaged on all structures belonging to 𝒟 𝒞 . Barnacle [ 20 ], which utilizes data-efficient machine learning, generally achieves higher precision than PyDCA, which relies on the pseudo-likelihood maximization direct coupling analysis approach as it is more effective in leveraging MSA information to predict contacts. Table 2 of Supplementary Information contains all the top L precision values. There is no clear trend between bit score and precision, as both tools demonstrate variability in performance across different bit scores. This is because not only the bit score but also the effective number of sequences in the RFAM family plays a significant role in the ability of methods to extract structural information from MSA [ 12 ]. 3.2 Use-case 2: Automated RNA structures curation for quick assessment of AlphaFold3 capabilities on RNA AlphaFold3 [ 18 ] shows remarkable promise in protein structure prediction and also promises to predict RNA and RNA complexes. Here, we assess its capabilities to predict the structure of unseen RNA sequences. To conduct such a comprehensive and unbiased assessment, we meticulously crafted two datasets using NucleoSeeker. Our first dataset, which we designated 𝒟 22 , comprised 213 RNA structures solved before 2023 as the training dataset used by AlphaFold3 was derived from the PDB accessed on January 2023. We applied stringent selection criteria to ensure the highest quality and representativeness of this dataset (for a detailed description of these criteria see Supplementary Information Section 3.2). Complementing this, we created a second dataset, 𝒟 23−24 , consisting of 27 structures solved in 2023 and 2024. This more recent dataset was important for assessing Alphafold3’s performance on newly determined structures. Since we know, that sequence identity is an important factor in reducing redundancy, we categorize the structures in 𝒟 23−24 based on their similarity to those in 𝒟 22 by calculating pairwise sequence identity (SI) between all structures across both datasets and dividing the 𝒟 23−24 entries in three subclasses according to their SI: below 50%, between 51% and 75% and above 76% (see Supplementary Table 3). In Fig. 1.c , we plot the pTM score, a confidence metric for the predicted structure from AlphaFold3 [ 18 ], against the RMSD between the predicted and experimental structure, for all structures in 𝒟 23−24 and according to their sequence identity with the closest match in the AF3 training se. Our quick assessment revealed that the AlphaFold3 confidence score genuinely provides a good indication of prediction quality, as high pTM scores correspond to RMSD values generally smaller than 5 Å. Additionally, the performance of AF3 improves for structures with higher sequence identity, as indicated by the green diamonds. This observation suggests that the model’s accuracy is influenced by the similarity between the target structure and those in AF3 training data (check section 3.2 of the Supplementary Information for more details). While Alphafold3 shows promising performance, especially for RNAs with some degree of similarity to known structures, this preliminary analysis shows that there is still room for improvement in predicting novel or highly divergent RNA structures. As we move forward, these insights will guide our efforts to refine and enhance RNA structure prediction methodologies. 4 Discussion NucleoSeeker targets to complement the development of efficient deep learning methods for RNA structure prediction, by providing a robust and flexible method for curating high-quality datasets from the PDB database. One of the key strengths of this software is its ability to apply a wide range of filters at both the structure and sequence levels, allowing researchers to create highly specific and relevant datasets tailored to their particular research needs. This functionality is particularly valuable given the challenges associated with RNA data, such as high redundancy, poor resolution, and the presence of hybrid structures. Moreover, the system is designed to ensure that datasets remain up-to-date, even in the rapidly evolving field of RNA research, where new structures are continually being determined and added to the database. Additionally, the modular design of NucleoSeeker allows its components to be used independently or in combination, providing researchers with a high degree of flexibility. Whether the goal is to filter specific structures, analyze polymer chains, or reduce dataset redundancy, it offers the tools needed to achieve these objectives efficiently in a simple way. 5 Data Availability The code for NucleoSeeker is available on GitHub at https://github.com/theuutkarsh/nucleoseeker and on Zenodo https://doi.org/10.5281/zenodo.13843170 . The package is also available through PyPi, Python Package Index. 6 Acknowledgments The authors gratefully acknowledge the Gauss Centre for Supercomputing e.V. ( www.gauss-centre.eu ) for funding this project by providing computing time through the John von Neumann Institute for Computing (NIC) on the GCS Supercomputer JUWELS at Jülich Supercomputing Centre (JSC) [ 21 ]. The authors gratefully acknowledge computing time on the supercomputer JURECA [ 22 ] at Forschungszentrum Jülich under the grant name EatsRNA. References [1]. ↵ Yann LeCun , Yoshua Bengio , and Geoffrey Hinton . “ Deep Learning ”. In: Nature 521 . 7553 ( May 2015 ), pp. 436 – 444 . issn: 1476-4687 . doi: 10.1038/nature14539 . OpenUrl CrossRef PubMed [2]. ↵ John Jumper et al. “ Highly Accurate Protein Structure Prediction with AlphaFold ”. In: Nature 596 . 7873 ( Aug . 2021 ), pp. 583 – 589 . issn: 1476-4687 . doi: 10.1038/s41586-021-03819-2 . OpenUrl CrossRef PubMed [3]. ↵ Martin Weigt et al. “ Identification of Direct Residue Contacts in Protein– Protein Interaction by Message Passing ”. In: Proceedings of the National Academy of Sciences 106 . 1 ( Jan . 2009 ), pp. 67 – 72 . doi: 10.1073/pnas.0805923106 . OpenUrl Abstract / FREE Full Text [4]. ↵ Alexander Schug et al. “ High-Resolution Protein Complexes from Integrating Genomic Information with Molecular Simulation ”. In: Proceedings of the National Academy of Sciences 106 . 52 ( Dec . 2009 ), pp. 22124 – 22129 . doi: 10.1073/pnas.0912100106 . OpenUrl Abstract / FREE Full Text [5]. ↵ Eleonora De Leonardis et al. “ Direct-Coupling Analysis of nucleotide coevolution facilitates RNA secondary and tertiary structure prediction ”. In: Nucleic acids research 43 . 21 ( 2015 ), pp. 10444 – 10455 . OpenUrl [6]. ↵ Fabrizio Pucci and Alexander Schug . “ Shedding Light on the Dark Matter of the Biomolecular Structural Universe: Progress in RNA 3D Structure Prediction ”. In: Methods. Experimental and Computational Techniques for Studying Structural Dynamics and Function of RNA 162–163 ( June 1, 2019 ), pp. 68 – 73 . issn: 1046-2023 . doi: 10.1016/j.ymeth.2019.04.012 . url: https://www.sciencedirect.com/science/article/pii/S1046202318303918 . OpenUrl CrossRef [7]. ↵ Helen M. Berman et al. “ The Protein Data Bank ”. In: Nucleic Acids Research 28 . 1 ( Jan . 2000 ), pp. 235 – 242 . issn: 0305-1048 . doi: 10.1093/nar/28.1.235 . OpenUrl CrossRef PubMed Web of Science [8]. ↵ Anne AH de Hond et al. “ Guidelines and quality criteria for artificial intelligence-based prediction models in healthcare: a scoping review ”. In: NPJ digital medicine 5 . 1 ( 2022 ), p. 2 . OpenUrl CrossRef [9]. ↵ Andrea Apicella , Francesco Isgrò , and Roberto Prevete . “ Don’t Push the Button! Exploring Data Leakage Risks in Machine Learning and Transfer Learning ”. In: arXiv preprint arxiv: 2401.13796 ( 2024 ). [10]. ↵ José Almeida Cruz et al. “ RNA-Puzzles: A CASP-like Evaluation of RNA Three-Dimensional Structure Prediction ”. In: RNA 18 . 4 ( Apr . 2012 ), pp. 610 – 625 . issn: 1355-8382, 1469-9001 . doi: 10.1261/rna.031054.111 . OpenUrl Abstract / FREE Full Text [11]. ↵ Rhiju Das et al. “ Assessment of Three-Dimensional RNA Structure Prediction in CASP15 ”. In: bioRxiv: The Preprint Server for Biology ( Oct . 2023 ), p. 2023.04.25.538330. issn: 2692-8205 . doi: 10.1101/2023.04.25.538330 . OpenUrl Abstract / FREE Full Text [12]. ↵ Fabrizio Pucci et al. “ Evaluating DCA-based Method Performances for RNA Contact Prediction by a Well-Curated Data Set ”. In: RNA 26 . 7 ( July 2020 ), pp. 794 – 802 . issn: 1355-8382, 1469-9001 . doi: 10.1261/rna.073809.119 . OpenUrl Abstract / FREE Full Text [13]. ↵ Bartosz Adamczyk , Maciej Antczak , and Marta Szachniuk . “ RNAsolo: A Repository of Cleaned PDB-derived RNA 3D Structures ”. In: Bioinformatics 38 . 14 ( July 2022 ), pp. 3668 – 3670 . issn: 1367-4803 . doi: 10.1093/bioinformatics/btac386 . OpenUrl CrossRef PubMed [14]. ↵ Eric P. Nawrocki and Sean R. Eddy . “ Infernal 1.1: 100-Fold Faster RNA Homology Searches ”. In: Bioinformatics 29 . 22 ( Nov . 2013 ), pp. 2933 – 2935 . issn: 1367-4803 . doi: 10.1093/bioinformatics/btt509 . OpenUrl CrossRef PubMed Web of Science [15]. ↵ Ioanna Kalvari et al. “ Rfam 14: expanded coverage of metagenomic, viral and microRNA families ”. In: Nucleic Acids Research 49 . D1 ( 2021 ), pp. D192 – D200 . OpenUrl CrossRef [16]. ↵ Fabian Sievers et al. “ Fast, Scalable Generation of High-quality Protein Multiple Sequence Alignments Using Clustal Omega ”. In: Molecular Systems Biology 7 . 1 ( Jan . 2011 ), p. 539 . issn: 1744-4292 . doi: 10.1038/msb.2011.75 . OpenUrl CrossRef PubMed Web of Science [17]. ↵ Peter Rice , Ian Longden , and Alan Bleasby . “ EMBOSS: The European Molecular Biology Open Software Suite ”. In: Trends in Genetics 16 . 6 ( June 2000 ), pp. 276 – 277 . issn: 0168-9525 . doi: 10.1016/S0168-9525(00)02024-2 . OpenUrl CrossRef PubMed Web of Science [18]. ↵ Josh Abramson et al. “ Accurate Structure Prediction of Biomolecular Interactions with AlphaFold 3 ”. In: Nature 630 . 8016 ( June 2024 ), pp. 493 – 500 . issn: 1476-4687 . doi: 10.1038/s41586-024-07487-w . OpenUrl CrossRef PubMed [19]. ↵ Mehari B Zerihun et al. “ Pydca v1.0: A Comprehensive Software for Direct Coupling Analysis of RNA and Protein Sequences ”. In: Bioinformatics 36 . 7 ( Apr . 2020 ), pp. 2264 – 2265 . issn: 1367-4803 . doi: 10.1093/bioinformatics/btz892 . OpenUrl CrossRef PubMed [20]. ↵ Oskar Taubert et al. “ RNA Contact Prediction by Data Efficient Deep Learning ”. In: Communications Biology 6 . 1 ( Sept . 2023 ), pp. 1 – 8 . issn: 23993642 . doi: 10.1038/s42003-023-05244-9 . OpenUrl CrossRef [21]. ↵ JSC . “ JUWELS Cluster and Booster: Exascale Pathfinder with Modular Supercomputing Architecture at Juelich Supercomputing Centre ”. In: Journal of Large-scale Research Facilities 7 ( 2021 ), A138 . doi: 10.17815/jlsrf-7-183 . OpenUrl CrossRef [22]. ↵ JSC . “ JURECA: Data Centric and Booster Modules implementing the Modular Supercomputing Architecture at Jülich Supercomputing Centre ”. In: Journal of large-scale research facilities 7 . A182 ( 2021 ). doi: 10.17815/jlsrf-7-182 . url: http://dx.doi.org/10.17815/jlsrf-7-182. OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted December 10, 2024. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following NucleoSeeker: Precision Filtering of RNA Databases to Curate High-Quality Datasets Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share NucleoSeeker: Precision Filtering of RNA Databases to Curate High-Quality Datasets Utkarsh Upadhyay , Fabrizio Pucci , Julian Herold , Alexander Schug bioRxiv 2024.12.06.626307; doi: https://doi.org/10.1101/2024.12.06.626307 Share This Article: Copy Citation Tools NucleoSeeker: Precision Filtering of RNA Databases to Curate High-Quality Datasets Utkarsh Upadhyay , Fabrizio Pucci , Julian Herold , Alexander Schug bioRxiv 2024.12.06.626307; doi: https://doi.org/10.1101/2024.12.06.626307 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7652) Biochemistry (17752) Bioengineering (13936) Bioinformatics (42084) Biophysics (21501) Cancer Biology (18655) Cell Biology (25586) Clinical Trials (138) Developmental Biology (13410) Ecology (19949) Epidemiology (2067) Evolutionary Biology (24378) Genetics (15639) Genomics (22562) Immunology (17779) Microbiology (40505) Molecular Biology (17219) Neuroscience (88825) Paleontology (667) Pathology (2845) Pharmacology and Toxicology (4840) Physiology (7666) Plant Biology (15182) Scientific Communication and Education (2048) Synthetic Biology (4305) Systems Biology (9840) Zoology (2274)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.