Taxonize-gb: A tool for filtering GenBank non-redundant databases based on taxonomy

preprint OA: closed CC-BY-NC-ND-4.0
📄 Open PDF Full text JSON View at publisher
AI-generated deep summary by claude@2026-07, 2026-07-04 · read from full text

This paper presents taxonize-gb, a command-line Python tool that filters NCBI GenBank non-redundant nucleotide and protein reference databases to retain only sequences associated with one or more specified taxonomy identifiers, enabling construction of taxa-specific offline databases for metagenomic analyses. The authors motivate the approach by noting that GenBank non-redundant resources grow rapidly and include many off-target references, making full-database searches computationally expensive, while existing alternatives are either not comprehensive for all marker genes or are not tailored to specific taxonomic groups. The key result is a practical software workflow that reduces search times by creating custom reference databases aligned to the taxa of interest. A major caveat highlighted is that, although NCBI offers an online experimental BLAST non-redundant database on a domain level, those sequences are not available for offline command-line download, which taxonize-gb addresses. This paper does not explicitly discuss endometriosis or adenomyosis; it was included in the corpus via a keyword match in the upstream search index.

Read from the paper's body, not the abstract. Not a substitute for reading the paper. No clinical advice. How this works

Abstract

Analyzing taxonomic diversity and identification in diverse ecological samples has become a crucial routine in various research and industrial fields. While DNA barcoding marker-gene approaches were once prevalent, the decreasing costs of next-generation sequencing have made metagenomic shotgun sequencing more popular and feasible. In contrast to DNA-barcoding, metagenomic shotgun sequencing offers possibilities for in-depth characterization of structural and functional diversity. However, analysis of such data is still considered a hurdle due to absence of taxa-specific databases. Here we present taxonize-gb, a command-line software tool to extract GenBank non-redundant nucleotide and protein databases, related to one or more input taxonomy identifier. Our tool allows the creation of taxa-specific reference databases tailored to specific research questions, which reduces search times and therefore represents a practical solution for researchers analyzing large metagenomic data on regular basis. Taxonize-gb is an open-source command-line Python-based tool freely available for installation at https://pypi.org/project/taxonize-gb/ and on GitHub https://github.com/msabrysarhan/taxonize_genbank . It is released under Creative Commons Attribution-NonCommercial 4.0 International License (CC BY-NC 4.0).
Full text 25,105 characters · extracted from preprint-html · click to expand
Taxonize-gb: A tool for filtering GenBank non-redundant databases based on taxonomy | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Taxonize-gb: A tool for filtering GenBank non-redundant databases based on taxonomy View ORCID Profile Mohamed S. Sarhan , View ORCID Profile Michele Filosi , View ORCID Profile Frank Maixner , View ORCID Profile Christian Fuchsberger doi: https://doi.org/10.1101/2024.03.22.586347 Mohamed S. Sarhan 1 Institute for Biomedicine, Eurac Research , Bolzano 39100, Italy ( Affiliated institute with Lübeck University , Lübeck, Germany ) 2 Department CIBIO, University of Trento , Trento 38123, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Mohamed S. Sarhan For correspondence: mohamed.sarhan{at}eurac.edu m.sabrysarhan{at}gmail.com Michele Filosi 1 Institute for Biomedicine, Eurac Research , Bolzano 39100, Italy ( Affiliated institute with Lübeck University , Lübeck, Germany ) Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michele Filosi Frank Maixner 3 Institute for Mummy Studies, Eurac Research , Bolzano 39100, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Frank Maixner For correspondence: mohamed.sarhan{at}eurac.edu m.sabrysarhan{at}gmail.com Christian Fuchsberger 1 Institute for Biomedicine, Eurac Research , Bolzano 39100, Italy ( Affiliated institute with Lübeck University , Lübeck, Germany ) Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Christian Fuchsberger For correspondence: mohamed.sarhan{at}eurac.edu m.sabrysarhan{at}gmail.com Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Analyzing taxonomic diversity and identification in diverse ecological samples has become a crucial routine in various research and industrial fields. While DNA barcoding marker-gene approaches were once prevalent, the decreasing costs of next-generation sequencing have made metagenomic shotgun sequencing more popular and feasible. In contrast to DNA-barcoding, metagenomic shotgun sequencing offers possibilities for in-depth characterization of structural and functional diversity. However, analysis of such data is still considered a hurdle due to absence of taxa-specific databases. Here we present taxonize-gb, a command-line software tool to extract GenBank non-redundant nucleotide and protein databases, related to one or more input taxonomy identifier. Our tool allows the creation of taxa-specific reference databases tailored to specific research questions, which reduces search times and therefore represents a practical solution for researchers analyzing large metagenomic data on regular basis. Taxonize-gb is an open-source command-line Python-based tool freely available for installation at https://pypi.org/project/taxonize-gb/ and on GitHub https://github.com/msabrysarhan/taxonize_genbank . It is released under Creative Commons Attribution-NonCommercial 4.0 International License (CC BY-NC 4.0). Introduction and motivation Environmental metabarcoding is a powerful molecular biology technique used to analyze the biodiversity within complex environmental samples ( 1 ). It operates by targeting and amplifying specific DNA regions, such as the 16S ribosomal RNA gene for bacteria, the COI gene for animals, or trnL for plants from a mixed sample of organisms. Once amplified, these genetic sequences are then subjected to high-throughput DNA sequencing, generating millions of short DNA sequences. Custom bioinformatic tools should be developed to in silico match these sequences to reference databases, allowing to identify and quantify the different species present in the original sample ( 2 ). Environmental metabarcoding has been used in different areas of research. For example, it has been used in food authentication ( 3 ), to understand plant-pollinator interactions ( 4 ), to detect invasive species in the environment ( 5 ), to monitor anthropogenic pollution ( 6 ), to reconstruct dietary components ( 7 ), and to reconstruct ancient ecosystems ( 8 ). Most of these studies relied on targeted amplicon approach, which employs amplification of single-marker genes to target specific taxa ( 9 , 10 ). However, choosing the appropriate marker gene for each taxon remains a perplexing issue, given the varying sensitivity and resolution levels of different markers. Selection of maker genes is also highly dependent on the quality of the reference database and availability of suitable unbiased universal primers, which ends up in a trade-off situation between feasible in vitro amplification and reliable in silico identification. Therefore, various studies suggested usage of multiplexed marker genes, which reported to be efficient in increasing species detection ( 11 , 12 ). However, such approach doubles the overall computational cost of the analysis which adds another factor to be considered in the trade-off. During the past decade, the costs of next-generation sequencing (NGS) have continued to decrease, making it more affordable for all biology disciplines. Such affordability encouraged environmental DNA researchers to shift towards the use of shotgun metagenomic sequencing instead of targeting only single or few marker gene amplicons. Shotgun metagenomic sequencing offers multiple advantages to the environmental DNA, such as targeting more genomic regions and avoiding the PCR bias-related issues. Using shotgun metagenomics is sometimes an indispensable approach particularly when analyzing very low biomass and low DNA samples, like in the cases of ancient DNA analysis or forensics, where the DNA is highly damaged which makes the retrieval of DNA amplicons a highly challenging task and amenable to many technical biases ( 13 ). While several curated databases have been already established to provide valuable resources for metabarcoding marker genes, a recurring challenge has been the lack of consistent maintenance and updates. These databases often start off as well-structured repositories of accurate and organized data, which over time with the rapid pace of new deposited data, the information contained within these databases become outdated. There are some well curated marker gene databases that are regularly updated and maintained ( 14 ), however they do not include all used metabarcoding marker genes. Therefore, there are new initiatives to develop database curation tools, such as Bcdatabaser ( 15 ) MetaCurator ( 16 ), which allow the user to develop up-to-date custom databases suitable for specific research questions and confined to particular taxonomic groups of interest. While the analysis of shotgun metagenomic data requires more comprehensive genome-wide databases, it also requires high computational resources and often specialized infrastructures, to handle such big data. Therefore, selection of targeted curated non-redundant and up-to-date databases is of paramount importance. Accordingly, the National Center for Biotechnology Information (NCBI) hosts GenBank, which contains a comprehensive collection of genetic sequence data, including DNA, RNA, and protein sequences submitted by researchers from around the globe ( 17 ). The NCBI offers up-to-date non-redundant protein and nucleotide databases ( 18 ), which seem to be the most suitable reference databases for analyzing shotgun sequences from environmental DNA samples ( 19 ). However, due to their comprehensiveness and regular updates, there are two major concerns. First, these databases grow exponentially every year which makes them difficult to maintain even with big computational infrastructures ( 18 ). Second, they contain a lot of off-target references which are impractical to keep in the search database, especially when the researcher is interested in specific taxonomic group. For example, if the researcher is interested in analyzing plant diversity, it would be a waste of resources to keep all non-plant proteins/nucleotides in the search database (e.g., animals, bacteria, phages, etc.). Using taxa specific databases as reference to analyze shotgun metagenomic sequences could help in detangling this issue. However, such specific databases are not offered by the GenBank nor by other genomic repositories. Although the GenBank is offering now an online experimental BLAST non-redundant nucleotide database on domain level (Eukaryotes, Prokaryotes, and Viruses), the sequences of these databases are not available for download for offline command line usage. This issue becomes more pronounced when dealing with eukaryotic diversity, since in contrast to bacterial and archaeal diversity analysis, there are not many tools which are optimized for their analysis. Software functionality We developed the software tool “taxonize-gb” as a command-line tool, developed in Python 3, designed to streamline the retrieval and filtering of data from the NCBI GenBank protein and nucleotide databases. The tool comprises various modules, with one specifically tailored for accessing the File Transfer Protocol (FTP) directories of the NCBI ( https://ftp.ncbi.nlm.nih.gov/ ) to retrieve the GenBank database files, i.e., nt/nr FASTA-formatted sequence files, mapping of accession numbers to taxonomy IDs, and the NCBI taxonomy database ( 20 ). The core module of our tool is “taxonize_gb” which is designed to streamline data extraction from the NCBI GenBank NR/NT databases based on a user-specified taxonomy IDs (TaxID). The module “taxonize-gb” performs filtering on the non-redundant protein/nucleotide databases of the GenBank based on a specified TaxID, which can be at any taxonomic level. The module performs the filtering on three main steps: ( 1 ) It employs the module DiGraph of NetworkX ( 21 ) to parse content of the “nodes.dmp” and the “names.dmp” files from the taxonomy database, representing them as a graph structure. Then, based on the user provided TaxID, it extracts all descendant TaxIDs (graph nodes) and outputs them as a data frame to store them along with their corresponding scientific names ( Figure 1 ). ( 2 ) The module filters the mapping files (i.e., accession to taxonomy ID) to retain only the accession numbers associated with the input TaxID and its descendants. ( 3 ) In the last step, it uses the Biopython modules ( 22 ) to parse the non-redundant FASTA sequences and perform a search within their headers to identify the filtered accession numbers, and optionally user-provided keywords ( Figure 1 ). The module is designed to take minimal and flexible inputs from the user – The user can provide paths for different database files; in case they are available in the local system. For example, to get all non-redundant Viridiplantae protein records, you can run the following command: Download figure Open in new tab Figure 1. Visual workflow for the “taxonize_gb” module for filtering the NCBI non-redundant protein and nucleotide databases. taxonize_gb --db nr –db_path nr.gz --prot_acc2taxid prot.accession2taxid.gz --pdb_acc2taxid pdb.accession2taxid.gz -- taxid 33090 --out Viridiplantae_nr/ While if the user does not provide any of the input databases, the latest version of the necessary databases will be downloaded automatically based on the provided mandatory input ‘--db’ option. For example, to get all non-redundant Viridiplantae protein records, the user can run the following command: taxonize_gb --db nr --taxid 33090 --out Viridiplantae_nr/ Additionally, the module can utilize the optional features of including keywords in the search to refine the filtering, e.g., to filter for specific gene/protein names or to filter for organellular genes/genomes. Detailed explanations on how to use the modules with further examples are available in the GitHub repository page ( https://github.com/msabrysarhan/taxonize_genbank ). The last module in our tool is “get_taxonomy” which is a utility script that uses the ete3 toolkit to retrieve taxonomic lineages of a given FASTA file. This module would be useful when the user is interested to make an overview on the taxonomic distribution of the filtered databases. Evaluation The alternative available option to perform taxa-specific search using the NCBI non-redundant protein database (to the best of our knowledge) is to use DIAMOND search tool against the complete NCBI-nr database, restricting the search to specific taxonomic IDs (using “--taxonlist” flag). To evaluate the performance, we explored the efficiency of different search approaches for querying NCBI non-redundant protein database (NCBI-nr). We compared two methods of DIAMOND search: against taxonized NCBI-nr databases and against complete NCBI-nr database with restricted TaxIDs. For the comparison, we targeted the following taxa and TaxIDs: 1) Chordata [taxid: 7711]; 2) Fungi [taxid: 4751]; and Viridiplantae [taxid: 33090]. We used the published metagenomic data from ancient paleofeces ( 23 ). For each search job, we used 16 CPUs and 50 Gb of RAM. Our results clearly demonstrate a substantial advantage in terms of time efficiency when employing DIAMOND search against taxonized GenBank databases ( Figure 2 ). The search times were significantly shorter when using taxonized databases (means 1.13 - 1.45 h), as opposed to the traditional complete database searches with TaxIDs restrictions (means 8.96 - 10.8 h). Download figure Open in new tab Figure 2: Performance comparison in terms of the runtimes (h) of DIAMOND search against taxonized NCBI-nr vs complete NCBI-nr with restricted TaxID search option. For further information on the used metagenomic samples, please refer to Maixner et al. (2021). The data are publicly available at ENA: PRJEB44507. Conclusion Taxonize-gb is a versatile, easy-to-use command-line tool that provides a comprehensive solution for researchers working with metagenomic data across multiple research disciplines by enabling efficient downloads and taxonomy-based filtering. This gives researchers the flexibility to focus their analyses on their area of interest. Finally, the reduced search times provided by using taxonized databases provide a practical and beneficial solution for researchers dealing with large metagenomic data and data-intensive research projects. Funding Information This work was supported by the Department of Innovation, Research and University of the Autonomous Province of Bolzano-South Tyrol (Italy). CF was supported partially by the National Institutes of Health [grant R01 HG009976]. MSS was supported by ONCOBIOME - European Union’s Horizon 2020 research and innovation programme under [grant 825410]. Conflict of interest The authors declare there is no conflict of interest. Acknowledgements We are grateful to the support of the Life Science Compute Cluster (LiSC) of the University of Vienna. We thank Mohamed R. Abdelfadeel of Leibniz-IGZ for testing the tool. Footnotes https://github.com/msabrysarhan/taxonize_genbank References 1. ↵ Rishan , S.T. , Kline , R.J. , Rahman , M.S.J.E.A. ( 2023 ) Applications of environmental DNA (eDNA) to detect subterranean and aquatic invasive species: A critical review on the challenges and limitations of eDNA metabarcoding . 100370 . 2. ↵ Ruppert , K.M. , Kline , R.J. , Rahman , M.S. ( 2019 ) Past, present, and future perspectives of environmental DNA (eDNA) metabarcoding: A systematic review in methods, monitoring, and applications of global eDNA . Global Ecology and Conservation , 17 , e00547 . OpenUrl 3. ↵ Rodríguez , M.d.S.T. , Vanhollebeke , J. , Derycke , S.J.F.C. ( 2023 ) Evaluation of DNA metabarcoding using Oxford Nanopore sequencing for authentication of mixed seafood products . 145 , 109388 . OpenUrl 4. ↵ Baksay , S. , Andalo , C. , Galop , D. , et al. ( 2022 ) Using Metabarcoding to Investigate the Strength of Plant-Pollinator Interactions From Surveys of Visits to DNA Sequences . 10 , 735588 . OpenUrl 5. ↵ Van Nynatten , A. , Gallage , K.S. , Lujan , N.K. , et al. ( 2023 ) Ichthyoplankton metabarcoding : An efficient tool for early detection of invasive species establishment . 6. ↵ Xu , X. , Yuan , Y. , Wang , Z. , et al. ( 2023 ) Environmental DNA metabarcoding reveals the impacts of anthropogenic pollution on multitrophic aquatic communities across an urban river of western China . 216 , 114512 . OpenUrl 7. ↵ Petrone , B.L. , Aqeel , A. , Jiang , S. , et al. ( 2023 ) Diversity of plant DNA in stool is linked to dietary quality, age, and household income . Proceedings of the National Academy of Sciences , 120 , e2304441120 . OpenUrl CrossRef 8. ↵ ter Schure , A.T.M. , Bruch , A.A. , Kandel , A.W. , et al. ( 2022 ) Sedimentary ancient DNA metabarcoding as a tool for assessing prehistoric plant use at the Upper Paleolithic cave site Aghitu-3, Armenia . Journal of Human Evolution , 172 , 103258 . OpenUrl 9. ↵ Hebert , P.D. , Cywinska , A. , Ball , S.L. , et al. ( 2003 ) Biological identifications through DNA barcodes . 270 , 313 – 321 . OpenUrl 10. ↵ Matiz-Ceron , L. , Reyes , A. , Anzola , J.J.F.i.P.S. ( 2022 ) Taxonomical evaluation of plant chloroplastic markers by bayesian classifier . 12 , 782663 . OpenUrl 11. ↵ Zhang , G.K. , Chain , F.J. , Abbott , C.L. , et al. ( 2018 ) Metabarcoding using multiplexed markers increases species detection in complex zooplankton communities . 11 , 1901 – 1914 . OpenUrl 12. ↵ Liu , J. , Zhang , H.J.F.i.M.S. ( 2021 ) Combining multiple markers in environmental DNA metabarcoding to assess deep-sea benthic biodiversity . 8 , 684955 . OpenUrl 13. ↵ Orlando , L. , Allaby , R. , Skoglund , P. , et al. ( 2021 ) Ancient DNA analysis . 1 , 14 . OpenUrl 14. ↵ Ratnasingham , S. , Hebert , P.D.J.M.e.n. ( 2007 ) BOLD: The Barcode of Life Data System ( http://www.barcodinglife.org ). 7 , 355 - 364 . OpenUrl 15. ↵ Keller , A. , Hohlfeld , S. , Kolter , A. , et al. ( 2020 ) BCdatabaser: on-the-fly reference database creation for (meta-)barcoding . Bioinformatics , 36 , 2630 – 2631 . OpenUrl CrossRef 16. ↵ Richardson , R.T. , Sponsler , D.B. , McMinn-Sauder , H. , et al. ( 2020 ) MetaCurator: A hidden Markov model-based toolkit for extracting and curating sequences from taxonomically-informative genetic markers . Methods in Ecology and Evolution , 11 , 181 – 186 . OpenUrl 17. ↵ Sayers , E.W. , Cavanaugh , M. , Clark , K. , et al. ( 2022 ) GenBank . 50 , D161 – D164 . OpenUrl 18. ↵ Pruitt , K.D. , Tatusova , T. , Maglott , D.R.J.N.a.r. ( 2005 ) NCBI Reference Sequence (RefSeq): a curated non-redundant sequence database of genomes, transcripts and proteins . 33 , D501 – D504 . OpenUrl 19. ↵ Xu , R. , Rajeev , S. , Salvador , L.C.J.P.o. ( 2023 ) The selection of software and database for metagenomics sequence analysis impacts the outcome of microbial profiling and pathogen detection . 18 , e0284031 . OpenUrl 20. ↵ Federhen , S. ( 2012 ) The NCBI Taxonomy database . Nucleic Acids Research , 40 , D136 – D143 . OpenUrl CrossRef PubMed Web of Science 21. ↵ Hagberg , A. , Swart , P. , S Chult , D. ( 2008 ) Exploring network structure, dynamics, and function using NetworkX . Los Alamos National Lab.(LANL), Los Alamos , NM (United States ). 22. ↵ Cock , P.J. , Antao , T. , Chang , J.T. , et al. ( 2009 ) Biopython: freely available Python tools for computational molecular biology and bioinformatics . 25 , 1422 . OpenUrl 23. ↵ Maixner , F. , Sarhan , M.S. , Huang , K.D. , et al. ( 2021 ) Hallstatt miners consumed blue cheese and beer during the Iron Age and retained a non-Westernized gut microbiome until the Baroque period . Current Biology , 31 , 5149 - 5162.e5146 . OpenUrl View the discussion thread. Back to top Previous Next Posted March 27, 2024. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Taxonize-gb: A tool for filtering GenBank non-redundant databases based on taxonomy Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Taxonize-gb: A tool for filtering GenBank non-redundant databases based on taxonomy Mohamed S. Sarhan , Michele Filosi , Frank Maixner , Christian Fuchsberger bioRxiv 2024.03.22.586347; doi: https://doi.org/10.1101/2024.03.22.586347 Share This Article: Copy Citation Tools Taxonize-gb: A tool for filtering GenBank non-redundant databases based on taxonomy Mohamed S. Sarhan , Michele Filosi , Frank Maixner , Christian Fuchsberger bioRxiv 2024.03.22.586347; doi: https://doi.org/10.1101/2024.03.22.586347 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7644) Biochemistry (17728) Bioengineering (13916) Bioinformatics (42037) Biophysics (21488) Cancer Biology (18636) Cell Biology (25552) Clinical Trials (138) Developmental Biology (13401) Ecology (19940) Epidemiology (2067) Evolutionary Biology (24367) Genetics (15621) Genomics (22545) Immunology (17764) Microbiology (40475) Molecular Biology (17208) Neuroscience (88744) Paleontology (667) Pathology (2842) Pharmacology and Toxicology (4834) Physiology (7659) Plant Biology (15175) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9834) Zoology (2272)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-06-06T02:00:05.402940+00:00
License: CC-BY-NC-ND-4.0