Full text
64,854 characters
· extracted from
preprint-html
· click to expand
A periodic table of bacteria?: Mapping bacterial diversity in trait space | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results A periodic table of bacteria?: Mapping bacterial diversity in trait space View ORCID Profile Michael Hoffert , View ORCID Profile Evan Gorman , Manuel E. Lladser , View ORCID Profile Noah Fierer doi: https://doi.org/10.1101/2025.07.11.664459 Michael Hoffert 1 Department of Ecology and Evolutionary Biology, University of Colorado Boulder , Boulder, CO USA 2 Cooperative Institute for Research in Environmental Sciences, University of Colorado Boulder , Boulder, CO USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michael Hoffert For correspondence: Michael.Hoffert{at}colorado.edu Noah.Fierer{at}colorado.edu Evan Gorman 3 Department of Applied Mathematics, University of Colorado Boulder , Boulder, CO USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Evan Gorman Manuel E. Lladser 3 Department of Applied Mathematics, University of Colorado Boulder , Boulder, CO USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Noah Fierer 1 Department of Ecology and Evolutionary Biology, University of Colorado Boulder , Boulder, CO USA 2 Cooperative Institute for Research in Environmental Sciences, University of Colorado Boulder , Boulder, CO USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Noah Fierer For correspondence: Michael.Hoffert{at}colorado.edu Noah.Fierer{at}colorado.edu Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Bacterial diversity can be overwhelming. There is an ever-expanding number of bacterial taxa being discovered, but many of these taxa remain uncharacterized with unknown traits and environmental preferences. This diversity makes it challenging to interpret ecological patterns in microbiomes and understand why individual taxa, or assemblages, may vary across space and time. While we can use information from the rapidly growing databases of bacterial genomes to infer traits, we still need an approach to organize what we know, or think we know, about bacterial taxa to match taxonomic and phylogenetic information to trait inferences. Inspired by the periodic table of the elements, we have constructed a ‘periodic table’ of bacterial taxa to organize and visualize monophyletic groups of bacteria based on the distributions of key traits predicted from genomic data. By analyzing 50,745 genomes across 31 bacterial phyla, we used the Haar-like wavelet transformation, a model-free transformation of trait data, to identify clades of bacteria which are nearly uniform with respect to six selected traits - oxygen tolerance, autotrophy, chlorophototrophy, maximum potential growth rate, GC content and genome size. The identified functionally uniform clades of bacteria are presented in a concise ‘periodic table’-like format to facilitate identification and exploration of bacterial lineages in trait space. While our approach could be improved and expanded in the future, we demonstrate its utility for integrating phylogenetic information with genome-derived trait values to improve our understanding of the bacterial diversity found in environmental and host-associated microbiomes. Introduction The periodic table of elements offers an interpretable framework for organizing atomic elements based on their properties [ 1 ]. This framework has been adapted across various fields to categorize the equivalent of elements, from food types to cell types, according to shared properties [ 2 – 5 ]. In fields like microbial ecology where organisms are the subjects of study, microbial taxa are documented in databases like the Genome Taxonomy Database [ 6 ] and BacDive [ 7 ] which serve to organize reference genomes, phylogenetic trees, and trait information. However, these databases omit a key feature of a periodic table: organizing units according to functional information. Microbial traits are as much the central focus of microbial ecology as chemical properties are the focus of physical chemistry: traits of microbes are analogous to chemical properties of atoms, where each determine the outcomes of small-scale interactions in their respective systems. Therefore, building a periodic table of microbial diversity that clearly organizes taxa and lineages based on shared traits would highlight key differences and illustrate patterns or periodicity in what we know, or think we know, about the diversity of bacterial life on Earth. Although applying this idea has been considered in some biological contexts (e.g. ecological niches in Winemiller et al., 2015), there is no consensus about how to interpretably measure, organize, or visualize the distributions of microbial traits across large swaths of taxonomic diversity. Establishing a framework for associating bacterial taxonomic identify with trait distributions would not only enhance our understanding of microbial ecology but also facilitate meaningful comparisons and insights across the rapidly expanding landscape of microbial research. Despite its promise, constructing a “periodic table of bacterial taxa” (PTBT) which succinctly visualizes bacterial trait-taxonomy relationships is challenging. Finding and assigning trait measurements to taxa in the PTBT is hampered by the intrinsic difficulty of measuring most bacterial traits, nebulous or dated taxonomic assignments [ 9 , 10 ], or complex trait distributions arising from horizontal gene transfer [ 11 ] and variable rates of evolution [ 12 ]. Although commonly-cited examples of broad taxonomic groups with conserved metabolisms exist, such as photoautotrophic Cyanobacteria [ 13 ], most bacterial traits are “distributed in phylogenetic clusters with a continuum of depths” [ 14 ] and cannot be uniformly attributed to a taxonomic group without empirical evidence. Establishing such empirical estimates of trait values for taxonomic groups is an evolving challenge. The traits of many bacterial taxa remain unknown because they are resistant to cultivation and not readily amenable to experimentation. However, cultivation-independent methods which infer trait values from genomic data [ 15 ] promise to circumvent these challenges and make it feasible to infer some traits from genomic data alone. As ongoing sequencing efforts yield thousands of new bacterial genomes per year, a combination of cultivation-independent and dependent methods have yielded databases of bacterial traits for an increasing diversity of bacteria (see Barberán et al., 2017; Madin et al., 2020; Reimer et al., 2019). A wide range of bacterial traits can now be inferred from genomic information, including autotrophy [ 18 – 20 ], nitrogen fixation [ 21 ], antibiotic resistance [ 22 ], maximum potential growth rates [ 23 ], temperature tolerances [ 24 – 26 ], pH preferences [ 27 ], and oxygen tolerances [ 28 ]. These advances have provided new means by which traits can be inferred for uncultivated and cultivated taxa alike, providing the empirical estimates of traits needed to build a PTBT. Here we show how a PTBT can be constructed using recently developed approaches to identify and visualize phylogenetic groups with particular trait distributions. Our approach addresses the challenges of finding trait information for most bacteria and interpreting the complex distribution of those traits in large phylogenies in two parts; first using genome-derived trait estimates to characterize a large swath of bacterial phylogenetic diversity and then applying the recently developed Haar-like wavelet projection (HWP) [ 29 ] to identify phylogenetic groups with conserved traits ( Figure 1 ). Here we focus on six key traits (genome size, GC content, oxygen tolerance, autotrophy, chlorophototrophy, and maximum potential growth rate), estimating each trait for over 50,000 representative genomes in the Genome Taxonomy Database (GTDB, Parks et al., 2018). We use the HWP to compute the trait variance associated with each phylogenetic node and identify clades with minimal trait variation - clades which can be accurately represented in single cells in a prototype periodic table ( Figure 1A,B ). Because many traits exhibit some degree of phylogenetic conservation [ 14 , 30 ], we expect that a small number of phylogenetic splits can capture large amounts of trait variance, and identification of these splits via HWP will make it feasible to design a PTBT simple enough to link taxa to traits while also accurately describing most of the variance in these traits ( Figure 1C-D ). Our approach represents an interpretable and quantifiable method to systematically characterize the largest sources of trait variance across the bacterial tree of life. Our use of quantitative means to discover and arrange the cells of the PTBT is deliberate, as we expect that this method can ultimately be expanded to add additional traits, improve inferences of individual trait values, and include additional taxa as genome-based models, experimental training data, and genomic datasets continue to improve. Download figure Open in new tab Figure 1. Conceptual outline of our method for constructing a periodic table of bacterial taxa (PTBT). The goal of a PTBT is to describe the distribution of traits across bacterial taxa for phylogenies with hundreds of thousands of leaves using quantitative methods. (A) We first apply the Haar-like wavelet projection (HWP) to a large, complex phylogeny with a trait measured for each leaf, e.g. a trait value predicted from a bacterial genome. The resulting Haar-like wavelet coefficients are associated with internal nodes of the phylogeny and measure the amount of variance uniquely associated with each node. (B) Our method of constructing a PTBT identifies the ancestral nodes of clades with small wavelet coefficients (green arrow), which are monophyletic groups with relatively uniform trait values. In contrast, clades with large wavelets (red arrow) cannot be accurately summarized by representing the collection of leaves with their average trait values. (C) The PTBT is constructed by collapsing clades with minimal variance in the trait to their ancestral node and using the tips of the collapsed phylogeny as “cells” in the table. The degree of collapsing is continuous: complete collapsing of the phylogeny (resulting in a single cell, tree I) is undesirable, as is the original, complex phylogeny (tree IV). An intermediate representation (trees II and III) illustrates which clades are functionally variable or uniform in a visually interpretable number of cells. (D) Therefore, the optimal PTBT selects the X largest wavelets to summarize trait-taxon relationships while preserving visual interpretability. Results Compilation of trait data for 50,745 bacterial genomes To accomplish the central task of arranging bacterial taxonomic in trait space, we first downloaded 62,291 species-level representative genomes from the Genome Taxonomy Database (GTDB) version 207 [ 6 ] for trait estimation. We used representative genomes only to ensure most genomes were relatively complete, uncontaminated, and had a standardized taxonomic assignment aligned with the GTDB phylogenetic tree. Our analyses were restricted to the 31 bacterial phyla which included at least 100 species-level representative genomes per phylum and only genomes with at least one ribosomal protein, a requirement for growth rate predictions [ 23 ], yielding 50,745 of the original 62,291 bacterial genomes. For each of these 50,745 genomes, we inferred values for six ecologically relevant traits: oxygen tolerance, autotrophy, chlorophototrophy, maximum potential growth rate, GC content, and genome size as described below and in the Methods section which includes full details on how these traits were determined from the genomic data. Supplementary Table 1 includes the inferred trait values for all 50,745 genomes. We emphasize that our trait inferences, particularly the inferences of autotrophy, phototrophy, and O 2 tolerance, are approximations which do not include all possible pathways for these functions and could ultimately be improved. However, as detailed below and in Figure 2 , our inferences broadly align with published literature, despite the literature’s bias toward taxa that are readily cultured and studied in vitro . Likewise, we describe the general patterns at broad taxonomic levels, acknowledging that the phylogenetic depth at which taxa exhibit consistency in traits varies depending on the trait and lineage in question (as discussed in more detail in the following sections). Download figure Open in new tab Figure 2. Distributions of estimated trait values for 50,745 GTDB representative genomes for six key traits. To ensure the data from all of these phyla are visible on this plot, the data are collapsed into bins represented by boxplots: each phylum is represented by at least one boxplot, where each boxplot shows a color-coded median and the 10 th ,25 th , 75 th , and 90 th quantile values for each trait across up to 500 genomes. Gray bands at the top of each plot act as a visual aid to delimit the boxes corresponding to each phylum using the phylogenetic placement of 31 bacterial phyla from GTDB v207 (Parks et al. 2022). Because autotrophy and phototrophy are binary variables, they are shown as bars representing the fraction of genomes in each bin that are positive for that trait. The six measured traits are shown in the following order: (1) oxygen tolerance, the probably of being aerobic, predicted using a random forest trained on BacDive genomes (2) maximum potential growth rate in estimated minimal doubling time, estimated with gRodon, (3) phototrophy, predicted using the presence of bacteriochlorophyll synthesis genes, (4) carbon fixation, predicted using the presence of key genes, (5) genome GC content and (6) genome size computed directly from genomic data. See Methods for details on how these traits were inferred. GC content and genome size were calculated directly from genomic data and included because these properties are associated with ecological attributes, from environmental tolerances to oxygen, light and temperature [ 31 – 34 ] to specific metabolisms [ 35 ] and ecological strategies [ 36 , 37 ]. Both genome size and GC content were considered as continuous trait values ranging from 0.22 – 25 Mbp and 15 -77%, respectively, across the 50,745 bacterial genomes. The taxa with very large genomes included many Actinobacteriota, Myxococcota, and specific clades of Proteobacteria and Cyanobacteria, while those with smaller genomes were found across many phyla - most Firmicutes, some Proteobacteria, and phyla like Omnitrophota and Patescibacteria where small genome sizes are a distinctive trait [ 37 , 38 ]. The genomes with the highest GC content were common in Actinobacteriota, Proteobacteria, and sister clades, while fewer and more specific subgroups of Firmicutes, Proteobacteria, and Bacteroidota contained low-GC genomes ( Figure 2 and Supplementary Table 1). Maximum potential growth rate was included because it is a key feature distinguishing variation in general life history strategies across bacteria [ 39 , 40 ] with maximum potential growth rate approximated using minimum potential doubling time predictions from gRodon v2 [ 41 ]. Values for maximum potential growth rate ranged from 0.01 to 25 hours, with lower values indicating faster potential growth and shorter generation times. Taxa with low estimated generation times (<1 hour) were relatively uncommon but phylogenetically widespread across anaerobic Firmicutes and aerobic Proteobacteria. Long estimated generation times (slowest maximum potential growth) were prevalent in many groups, including specific Proteobacteria, many Actinobacteriota, and Gemmatimonadota ( Figure 2 ). Taxa that have been isolated in vitro almost always have shorter predicted generation times (faster maximum potential growth rates) than corresponding, uncultivated taxa within the same phylum from which genomes were obtained via assembly from metagenomes (Supplementary Figure 8), highlighting a bias for faster growth among taxa that are readily cultivable [ 23 ]. The potential for autotrophic metabolism, or the ability to fix CO 2 , was estimated using key enzymes and pathway completeness [ 18 , 42 ] from two of the four known bacterial autotrophy pathways (Calvin Cycle and hydroxypropionate bi-cycle) which are reasonably well-characterized. This trait is encoded as a binary variable (either inferred to be capable or incapable of autotrophic CO 2 fixation) when a genome contained sufficient completeness of either pathway. Using this approach, 9.3% (4,627) of the bacterial genomes were inferred to be capable of autotrophic metabolism via either of the two measured pathways. As expected, the inferred capacity for autotrophic CO 2 fixation was most prevalent among Cyanobacteria and Proteobacteria, and otherwise found in small numbers of taxa in Actinobacteriota [ 43 ] sister phyla of the Firmicutes supergroup [ 44 – 47 ] ( Figure 2 ). Similarly, we screened all genomes for key genes and pathways to infer chlorophototrophy, the capacity to capture light energy using chlorophyll. Approximately 6% (3,223) of bacterial genomes contained chlorophyll synthesis genes ubiquitous in a genomic dataset of chlorophototrophs [ 20 ]. Consistent with the literature [13, 20, 48–50], the inferred capacity for chlorophototrophy was largely (but not always) restricted to taxa within the phyla Proteobacteria, Cyanobacteria, Chloroflexota, and other groups which were also inferred to be autotrophic ( Figure 2 ) Finally, oxygen tolerance was predicted from gene presence using a random forest model trained on characterized taxa from BacDive [ 7 ]. Oxygen tolerance was encoded as the probability each genome is aerobic (zero to one) and was included because tolerance of oxygen is a key determinant of metabolic strategies [ 51 ] and the potential environments in which taxa can thrive. As expected, oxygen tolerance is not strongly conserved at the phylum level ( Figure 2 ). Aerobes are particularly prevalent in Proteobacteria, Actinobacteriota, and at least some clades of most other phyla, while taxa within Firmicutes A, Desulfobacterota, and subgroups of some Bacteroidota were almost exclusively inferred to be anaerobes. Identification of taxa with shared traits using Haar-like wavelets To identify lineages with conserved traits across all 50,745 bacterial genomes (from the genome-derived inferences described above), we summarized trait-phylogeny relationships using the Haar-like wavelet projection (HWP). The HWP performs phylogenetically-informed analyses of trait values associated with species, or leaves, into measurements of the trait variation uniquely attributed to each interior node of the phylogeny (see Gorman & Lladser, 2024 for specific methods). The magnitude of a “wavelet coefficient” associated with each node quantifies the degree of change observed between the leaves descended from any given node in the phylogenetic tree. The wavelet coefficients normalized to fraction of total variance are shown in Supplementary Figure 2 and reveal that most trait variance is explained by only a few internal nodes. For example, at least 25% of the variation in oxygen tolerance, growth rate, GC content, and genome size are attributed to one split (clade c000130) at the common ancestor of slow-growing high GC aerobes with large genomes (Actinobacteriota) and fast-growing low-GC anaerobes with small genomes (most Firmicutes). Fewer than 100 wavelets are required to explain 60% or more of total trait variance ( Figure 3 ); 26 of these wavelets are shared across two or more traits, implying that trait changes frequently occur at the same phylogenetic nodes, perhaps due to coordinated evolution of traits [ 14 ]. Additionally, many of the phylogenetic splits with large wavelet coefficients reside at deeper nodes, often at approximately the phylum level of resolution (Supplementary Figure 3). Collectively the HWP reveals that these six traits are correlated, significant sources of variation are found deep in the phylogeny, and small numbers of splits can describe trait distributions accurately, all observations which indicate that a PTBT can describe a majority of the trait variance across a broad diversity of bacteria, even when considering multiple traits simultaneously. Download figure Open in new tab Figure 3. Per-coefficient variance explained for each trait versus the number of wavelets used to describe that variance. Each curve is constructed by sorting the wavelet coefficients for each trait by magnitude and summing the magnitudes. The intersection of horizontal bars at 10% variance intervals indicate the total number of unique wavelet coefficients (or phylogenetic splits) which explain that degree of total variance for the trait. For the periodic table, we selected a threshold of 60% variance explained, which yielded a total of 1307 ‘cells’, or groupings of bacteria, with the 256 cells which contained the largest number of unique genomes (∼80% out of the 50,745 genomes) shown in Figure 5, our ‘periodic table’. Association of traits with one another and phylogenetic structure is a desirable condition given that the utility of the PTBT is tied to its visual complexity and information fidelity. The deconvolution of phylogeny-trait relationships provided by the HWP illustrate that trait distributions and co-occurrence are not combinatorially complex, and that trait-taxon relationships can be simplified while preserving variation in a PTBT. Other studies have also determined that vertical transmission and deep conservation are properties of complex bacterial traits [ 14 , 30 ], but our data reveal that these trends are not expressed equally among the traits examined here. The small number of highly explanatory wavelets ( Figure 3 and SF2) in the autotrophy and growth rate traits, and to lesser extents in phototrophy and genome size, indicate these traits have relatively lower degrees of phylogenetic conservation (and may be more difficult to summarize). The different degrees of conservation in genome size and GC content, measured directly from genomes, indicate that variation in phylogenetic conservation is biological and not simply due to inference errors. Among our traits, various explanations may explain varying degrees of phylogenetic association, from a history of horizontal transmission in the Calvin Cycle [ 18 ], to selection or drift altering genome size based on the specific biotic and abiotic factors present [ 36 , 52 – 54 ]. Such variations make visual properties which depict trait co-occurrence and depth of phylogenetic conservation important components of the PTBT’s visual design. Drafting a ‘periodic table’ to visualize bacterial diversity in trait space The procedure for constructing the PTBT involves analyzing the wavelet coefficients to find parents of sister clades whose leaves are not similar with respect to each trait and therefore do not represent the functionally uniform (or nearly uniform) clades that we want the cells of the PTBT represent. At least 60% of trait variance can be explained by only 72 unique wavelet coefficients and corresponding nodes ( Figure 3 ). The PTBT is constructed to explain a particular amount of variance in the data: the amount of trait variance explained (TVE) is arbitrary but determines the visual complexity of the PTBT, so we picked a value near the inflection point of the wavelet magnitude versus variance explained distribution, maximizing TVE while minimizing number of cells in the PTBT. The 72 unique nodes required to explain 60% of the variance were then used to prune branches and create a simplified phylogeny containing junctions which contribute to TVE. This pruned tree’s new leaves are clades which contain no large wavelets and therefore capture minimal descendent trait variation to the extent guaranteed by the 60% TVE threshold and are drawn in the PTBT ( Figure 5 ). A layout of each clade in two dimensions to emulate the draft periodic table was created using each clade’s pairwise phylogenetic distance and median trait values, and computing a t-SNE-based embedding of these relationships into a common three-dimensional space with ENS-t-SNE [ 55 ]. The resulting layout combines phylogenetic information with trait information to gather cells which are similar based on phylogenetic and/or trait-based distance ( Figure 4 ). The median trait values for each clade are drawn with different graphical elements of the cells in Figure 5A , our initial design of an empirically derived PTBT, including a diagram of the phylogeny which resulted from HWP-based collapsing to illustrate the taxonomic groups which constitute the cells in the table ( Figure 5B ). Download figure Open in new tab Figure 4. Data used to generate a layout of the periodic table diagram which combines trait and phylogenetic information using ENS-t-SNE. Each plot shows the distribution of six traits among the cells of the periodic table, with an additional plot to illustrate the phylum-level identities of the taxa within each of the cells. Each cell corresponds to a clade in the bacterial trait dataset (50,745 leaves total, 272 in this diagram) discovered by traversing the phylogeny until no sufficiently large Haar-like wavelets existed in any subtree. The median trait values from the resulting tips and a patristic distance matrix were provided to the ENS-t-SNE algorithm to generate a layout that included phylogenetic and trait information. We note that the cells and their arrangement in these panels are identical to those shown in the final periodic table shown in Figure 5, with this figure serving as a complement to Figure 5 to help visualize how traits and taxonomic identities vary across the cells in the periodic table. Download figure Open in new tab Figure 5. Periodic table of bacterial diversity. Functionally conserved taxonomic groups were identified by pruning the branches of a bacterial phylogeny with 50,745 leaves (species-level representative genomes) until an internal node associated with significant internal variance was discovered, essentially identifying the clades to put in a “periodic table-like” diagram. The layout of the table was generated by projecting a combination of trait and phylogenetic information into 2D space using ENS-t-SNE, then snapping points to a grid which preserved the nearby placement of clades with similar trait values and phylogenetic identity (see Figure 4). (A) A multi-trait illustration of the trait values for 272 groups of functionally homogeneous bacterial taxa discovered by the pruning algorithm, each illustrated as a single “cell.” The periodic table is subdivided into 8 parts for the purposes of description and the higher level categorization of bacterial diversity in trait space. (B) The underlying pruned phylogeny of bacteria labelling the clades which appear in the diagram from panel A (numbered, black edges), as well as some of the larger phylogenetic groups which divide the clades of bacterial taxa with homogeneous, or nearly homogeneous, trait values. The numbers indicate the regions of the periodic table (from panel A) in which some of the clades appear. The PTBT illustrates the distributions of traits across bacterial phylogenetic diversity and facilitates the interpretation of trait co-occurrence, conservation, and divergence patterns in specific taxa. First, the PTBT illustrates the variation in the phylogenetic depths at which trait combinations are conserved. The PTBT assigns Proteobacteria, Firmicutes A, and Actinobacteriota more cells (172, 24, and 21 respectively) because traits vary more in these phyla versus Myxococcota, Cyanobacteria, Spirochaetota, Patescibacteria, and other phyla with less variation in the traits assessed here ( Figure 4 ). Second, the layout identifies phylogenetically distinct taxa that share similar traits, separating primarily oxygen-tolerant Actinobacteriota with high GC contents and large-genomes in regions 3-4 of Figure 5A from phylogenetically distinct but otherwise similar pseudomonads in region 5 of Figure 5A . The PTBT also illustrates where related taxa have distinct traits, as in the placement of oxygen-tolerating photoautotrophic Cyanobacteriota ( Figure 5A region 2) and chemoheterotrophic Bacteroidota ( Figure 5A -8) that are far from each group’s single clade of chemoheterotrophic low-GC anaerobes (Bacteroidales in Bacteroidota; Vampirovibrionia in Cyanobacteria, both in region 6 of Figure 5A ). Finally, the PTBT makes it possible to identify the taxonomic underpinnings of high-level associations between traits, for example in the association between GC content, oxygen tolerance, and growth rates appearing to be driven here by the distinction between Firmicutes and Actinobacteria/Proteobacteria, a result consistent with previous work [ 52 ]. Essentially, the PTBT maps trait distributions into visual channels like distance, clustering, size, and color which enables users to rapidly explore bacteria with particular groups of traits, identify the set of traits any bacterial group is likely to have, and assess the relative diversity of particular groups in trait space by examining both the number of cells in a group and their placement relative to one another. Discussion This work explores how quantitative methods can help describe, explore, and visualize complex bacterial taxon-trait relationships. Inspired by the long-standing desire among microbial ecologists to find methods which identify and organize taxonomic groups with shared traits [56– 59] and recognizing that traits fundamentally determine organismal fitness and ecologies [ 60 ], we believe traits represent the most appropriate foundation on which to build a “periodic table” of bacterial diversity. Our methods of constructing a periodic table includes novel applications of genome-based trait inference methods, the HWP, and ENS-t-SNE to address persistent challenges in estimating traits across a broad diversity of bacteria and then identifying and organizing groups of bacteria into a visual framework that captures trait co-occurrence, phylogenetic conservation, and phylogenetic similarity. We can effectively organize large swaths of bacterial diversity in trait space with the PTBT because there are biological and evolutionary constraints that limit the possible combinations of trait states and allow an informative picture of bacterial taxon-trait relationships to be summarized and visualized. We expect that our approach could generate meaningful simplifications for other trait-taxon associations, with particular methods like the HWP providing interpretable information to inform the design of new representations of trait-phylogeny datasets. With adaptation, these methods could also help integrate trait data with genome databases like GTDB [ 6 ] to unify microbial functional, phylogenetic, and genomic information in a single interface. Although efforts to synthesize traits, phylogenetic information, and visualization techniques have many realized and unrealized benefits, we acknowledge that our workflow is inherently reductive and could be improved. We have only visualized six traits with a specified degree of variance, largely ignored potential errors in trait estimation methods, and made other simplifications. In addition to improvements to trait estimations and phylogenetic inferences, whether the PTBT provides verifiable explanations of the realized patterns in ecosystems should be critically assessed in further work. In the meantime, we hope the PTBT will inspire development of new tools to summarize, assess, and validate our current knowledge - and assumptions about - the taxon-trait associations upon which many analyses in microbial ecology explicitly or implicitly rely. For example, the PTBT illustrates where assumptions can and cannot be made about the depth and variability of trait conservation and co-occurrence, a topic brought under increasing scrutiny by the assumptions employed in metagenomic profiling tools [ 61 ]. Therefore, efficiently finding and presenting evidence for a particular trait’s presence in taxa – the essential tasks the PTBT is designed to do - serve important roles in microbial ecology, and our work provides an initial demonstration of how trait organization schemes may be designed and could inform evidence-based exploration of complex trait distributions across phylogenetically diverse and poorly characterized microbial communities. Materials and Methods Compilation of genomic data 62,291 representative genomes from GTDB release 207 ( https://data.gtdb.ecogenomic.org/releases/release207/207.0/ ) were downloaded for annotation with six traits, as described below. We restricted our analyses only to those genomes from phyla with 100 or more representatives. To ensure that maximum estimated growth rate could be included for all genomes, only those containing at least one ribosomal protein were included in the analyses. The remaining 50,745 genomes were used for subsequent analyses. Analyses and figures were generated using Python v3.10.10. Details on the 31 phyla represented and the number of genomes per phylum are provided in Supplemental Figure 1. GC content and genome size GC content and genome size were drawn directly from GTDB genome statistics, as calculated by CheckM [ 62 ]. Oxygen tolerance A list of NCBI taxon IDs in BacDive (accessed 6/9/2022) labeled as either ‘aerobe’ or ‘anaerobe’ were used to construct an oxygen tolerance dataset. Seven original BacDive oxygen tolerance labels from 6,629 genomes were grouped into two categories using the following scheme: anaerobe (22.2%), obligate anaerobe (1.5%), microaerophile (10.6%), facultative aerobe (0.8%), aerotolerant (0.08%), and microaerotolerant (0.03%) were assigned to ‘anaerobe’. Aerobe (46.1%), obligate aerobe (2.2%), and facultative anaerobe (8.15%) were assigned to ‘aerobe’. Each NCBI Taxon ID was matched to a representative GenBank assembly and corresponding GTDB representative genome. If no GenBank assembly was labeled as representative, we selected the highest-quality genome based on GTDB’s contamination and completeness estimates and used its GTDB species representative. When GTDB species clusters contained multiple oxygen annotations, the most common label for each cluster was paired to the respective genome. Statistics of the assembled dataset are available in the code resources (see Data Availability). 6,629 genomes with labels were identified, 662 used in final testing and 5,964 in cross-validation procedures. A random forest model implemented in sklearn v1.0.2 was used to predict oxygen tolerance. Protein families from Pfam present in each genome were used as predictors with oxygen tolerance categories (from BacDive, Reimer et al., 2019) used as the target variable. The predicted coding sequences from representative genomes were annotated using hmmscan (HMMER v.3.3.2) and PFam release 35.0 ( pfam.xfam.org ) on open reading frames (ORFs) identified with Prodigal v2.6.3 [ 63 ], filtering to hits with bitscores better less than the respective Pfam entry’s “trusted cutoff.” After all matching Pfam families per gene were determined, a presence/absence table was constructed with all 62,291 genomes in release 207 (later filtered to 50,745, as above) and 17,422 (out of 19,632) observed Pfam entries. A bootstrapped logistic regression with an L1 optimizer implemented in sklearn v1.0.2 [ 64 ] was used with 100 replicates and a lambda of 1 to reduce the number of genes in the input data. The fraction of bootstraps with non-zero regression coefficients for each gene was included in later hyperparameter tuning for random forests. We withheld 10% of data for model testing (herafter “testing data”) and retained 90% (5,964) of the 6,629 genomes in the training set for hyperparameter tuning using nested cross-validation, with 10-fold cross-validation for models and 5-fold cross-validation for combinations of parameters. We tuned the following parameters: 1) number of trees, 2) maximum tree depth, 3) minimum samples per leaf, 4) minimum samples per split, 5) bootstrapping, 6) number of features used to train each tree, and 7) number of bootstrapped logistic L1 regressions with a nonzero coefficient. Maximum training accuracy was achieved using Pfam entries with non-zero coefficients in 50-70% of randomized logistic regressions and more than 75 trees, but otherwise models were not sensitive to parameter choice. The final model used the following parameters to reduce risk of overfitting: Pfam entries present in 70% of bootstrapped logistic regressions, n_estimators = 3000, max_features = sqrt, max_depth = 13, min_samples_split=2, min_samples_leaf = 1, bootstrap = False. The random forest model achieved 92% cross-validation training accuracy and 91.4% test accuracy with modest differences in performance for aerobes and anaerobes (Supplementary Figure 4), likely due to unbalanced training data and better predictive power for aerobe-associated enzymes [ 28 ]. The model successfully recapitulated the original BacDive labels in the classification probability space (Supplementary Figure 5) and used enzymes associated with oxygen-dependent and independent metabolisms (Supplementary Figure 6). Inferring chlorophototrophy Due to difficulties distinguishing rhodopsin from bacteriorhodopsin, we have focused on chlorophototrophic organisms in this study. To identify proteins and pathways which indicate phototrophy, we identified the following chlorophototrophy-related GO terms: GO:0015995, GO:1902326, GO:0036068, GO:00333005, GO:0030494, GO:0010380, GO:1902325, GO:0036067, GO:0015979, GO:0019684, GO:0019685, GO:0010109, GO:1905157, GO:0009521, GO:1905156, GO:0034357. These GO terms were used to identify 48 Pfam families of photosynthesis-related proteins, listed in Supplementary Table 3. The GenBank assembly accessions for known chlorophototrophs from Thiel et al. (2018) were used to identify which phototrophy-related Pfam families were common among phototrophic taxa. Among 698 GTDB representative genomes identified as phototrophs, 24 had none of the Pfam families. The Pfams for Photosynthetic reaction center protein (PF00124), magnesium-protoporphyrin IX methyltansferase (PF07109), and proto-chlorophyllide reductase (PF08369) were the most prevalent, indicating the bacteriochlorophyll biosynthetic mechanism can be used to identify putative phototrophs. 95.7% of the 698 genomes in Thiel et al. 2018 contained at least one complete (100% GapSeq completeness) biosynthetic process for chlorophyll a. Genomes were pre-screened using photosynthesis-related Pfam families: 12,906 of the 62,291 GTDB representative contained at least one. These 12,906 genomes were scanned with Gapseq v1.2 [ 65 ] for chlorophyll photosynthesis capacity. Any genome with at least one chlorophyll-biosynthesis-related process that was more than 90% complete was considered phototrophic for downstream analyses. Inferring carbon fixation Carbon fixation potential was determined by analyzing genomes for genes associated with known carbon fixation pathways from previous literature and existing databases (Caspi et al., 2020; Kanehisa & Goto, 2000; Momper et al., 2017). Carbon fixation pathways such as the reductive tricarboxylic acid (rTCA) cycle can be difficult to distinguish from their oxidative variants using pathway completeness, so we first screened genomes for key enzymes using the HMMs from Asplund-Samuelsson and Hudson, 2021. After pre-screening, 10,166 genomes were analyzed with GapSeq to measure pathway completeness of the following MetaCyc pathways: CALVIN-PWY / Calvin Cycle, CODH-PWY / rAcoA homoacetogenic, PWY-7784 / rAcoA methanogenic, P23-PWY / rTCA I, PWY-5392 / rTCA II, PWY-5789 / 3HP/4HB, PWY-5743 / 3HP bicycle. Based on Asplund-Samuelson et al. 2021, the following completeness thresholds were used to identify putative autotrophs: CALVIN-PWY 90, CODH-PWY 95, PWY-7784 90, P23-PWY 90, PWY-5392 80, PWY-5789 90, PWY-5743 82. Using these cutoffs, 6,511 genomes out of 50,745 were identified with sufficient completeness of carbon fixation initially; only the 4,627 genomes containing Calvin Cycle or 3HP bi-cycle were considered autotrophic for the construction of the periodic table because other pathways were found in organisms not reported in the literature to be capable of carbon fixation. Maximum potential growth rate estimates To estimate maximum predicted doubling times for each genome, ribosomal proteins were annotated and processed using gRodon v2 [ 41 ], a codon usage bias (CUB) based method for estimating microbial doubling time. Ribosomal proteins were used as the highly expressed gene set (from which gRodon estimates growth rates) for each genome. To annotate ribosomal proteins, we used BLASTP v2.5.0 [ 66 ] to align predicted ORFs against the growthpred database of microbial ribosomal proteins (Vieira-Silva & Rocha, 2010). ORFs with at least 50% coverage and an e-value of 1e-5 were labeled as ribosomal. 3,053 genomes had no hits. Of the remaining 50,771 genomes, 29,642 had fewer than 10 annotated ribosomal genes (Supplementary Figure 7) and taxa with lower growth rates were biased toward isolate (vs. metagenome-assembled genome) sources. Wavelet projection and construction of the periodic table All subsequent analyses and figures were generated using Python v3.10.10 and ETE toolkit v3.1.2 ( https://github.com/etetoolkit/ete ). The GTDB r207 phylogeny was pruned using the ete3 “prune” function with keep_branch_lengths=True to contain quality-controlled species with estimates for all six traits (50,745 total). A Haar-like wavelet basis was computed for this tree with code from Gorman and Lladser, 2022 ( https://github.com/edgor17/Sparsify_Ultrametric ) and used to compute a 50,744 × 6 matrix of Harr-like wavelets coefficients per trait for each non-root interior node. The vector of Haar-like wavelet coefficients for each trait were divided by the L2 norm, yielding a normalized vector of variances explained per wavelet coefficient/ internal node. To identify nodes which would define a periodic table composed of functionally uniform groups, the normalized wavelet coefficients were sorted and cumulatively summed starting with the largest wavelet until a selected amount of variance was captured (60%) for each trait. Subtrees that did not contain any of the 72 total nodes identified by this procedure were collapsed using a custom tree-pruning algorithm. Algorithm to construct a collapsed tree T W : Define a set of nodes W to avoid collapsing Begin with the original phylogeny T Traverse T Starting with the root of T ; For node N in T: If N is a leaf of T : mark it as a leaf of T W . Else if all descendants of N are not in W : mark N as a leaf of T W . Else: Continue Build T W using only marked nodes as leaves; collapse all unmarked subtrees. This algorithm retains nodes only if they belonged to a wavelet clade or were necessary to maintain the tree structure connecting retained nodes. This algorithm pruned the original 50,745 tips to 272 tips (clades from the original phylogeny). Re-computing the trait data from the wavelet coefficients in this pruned tree would create a trait distribution that captured 60% or more of the variation from the original data. These 272 clades were used to draw the periodic table. A 272 × 272 patristic distance matrix of these clades and a 272 × 6 matrix of the scaled median trait values for the leaves associated with each clade was used to compute a three-dimensional ENS-t-SNE embedding with 4000 iterations and a perplexity of 50 using mview ( https://github.com/enggiqbal/MPSE-TSNE ). The images from the resulting embedding were combined and points were snapped to a grid using the Hungarian algorithm implemented in sklearn. The grid layout, median trait values for each clade, and phylogenetic tree were used to generate the periodic table diagram and tree diagrams in Figures 4 and 5 using custom Python scripts and the seaborn [ 67 ], numpy [ 68 ] and matplotlib [ 69 ] libraries. Author contributions M.H. and N.F. conceived this study with all analyses conducted by M.H. with E.G. and M.L. helping with the Haar-like wavelet analyses. Writing was led by M.H. and N.F. with reviewing and editing of the manuscript by E.G. and M.L. All authors read and approved the final manuscript. Conflicts of interest The authors declare no conflicts of interest. Funding This work was supported by grants from the US National Science Foundation (DEB 2126106, AW5809-826664, OPP 2133684) and funding provided to the Center for Microbial Exploration by the University of Colorado Boulder. Data Availability Intermediate data files and code used to generate figures and analyses are available on GitHub: https://github.com/realmichaelhoffert/bacterial_periodic_table/tree/main Acknowledgements We want to thank JL Weissman and members of the Fierer Lab for their insights and willingness to engage in productive discussions about this work. We also want to thank the Information Technology team at CIRES for computational support. We are deeply indebted to the team responsible for the Genome Taxonomy Database (GTDB) which made this work possible. Funder Information Declared National Science Foundation , DEB 2126106 , AW5809-826664 , OPP 2133684 Center for Microbial Exploration, University of Colorado Boulder References 1. ↵ Schwerdtfeger P , Smits OR , Pyykkö P. The periodic table and the physics that drives it . Nat Rev Chem 2020 ; 4 : 359 – 380 . doi: 10.1038/s41570-020-0195-y OpenUrl CrossRef PubMed 2. ↵ Gänzle M. The periodic table of fermented foods: limitations and opportunities . Appl Microbiol Biotechnol 2022 ; 106 : 2815 – 2826 . doi: 10.1007/s00253-022-11909-y OpenUrl CrossRef 3. J. Blower P. A nuclear chocolate box: the periodic table of nuclear medicine . Dalton Trans 2015 ; 44 : 4819 – 4844 . doi: 10.1039/C4DT02846E OpenUrl CrossRef PubMed 4. Lengler R , Eppler MJ . Towards a periodic table of visualization methods of management . Proc. IASTED Int. Conf. Graph. Vis. Eng. 2007 . USA : ACTA Press , 2007 , pp 83 – 88 . 5. ↵ Xia B , Yanai I. A periodic table of cell types . Development 2019 ; 146 : dev169854 . doi: 10.1242/dev.169854 OpenUrl Abstract / FREE Full Text 6. ↵ Parks DH et al. GTDB: an ongoing census of bacterial and archaeal diversity through a phylogenetically consistent, rank normalized and complete genome-based taxonomy . Nucleic Acids Res 2022 ; 50 : D785 – D794 . doi: 10.1093/nar/gkab776 OpenUrl CrossRef PubMed 7. ↵ Reimer LC et al. BacDive in 2019: bacterial phenotypic data for High-throughput biodiversity analysis . Nucleic Acids Res 2019 ; 47 : D631 – D636 . doi: 10.1093/nar/gky879 OpenUrl CrossRef PubMed 8. Winemiller KO et al. Functional traits, convergent evolution, and periodic tables of niches . Ecol Lett 2015 ; 18 : 737 – 751 . doi: 10.1111/ele.12462 OpenUrl CrossRef PubMed 9. ↵ Garrity GM . A New Genomics-Driven Taxonomy of Bacteria and Archaea: Are We There Yet? J Clin Microbiol 2016 ; 54 : 1956 – 1963 . doi: 10.1128/JCM.00200-16 OpenUrl Abstract / FREE Full Text 10. ↵ Godfray HCJ . Challenges for taxonomy . Nature 2002 ; 417 : 17 – 19 . doi: 10.1038/417017a OpenUrl CrossRef PubMed Web of Science 11. ↵ Thomas CM , Nielsen KM . Mechanisms of, and Barriers to, Horizontal Gene Transfer between Bacteria . Nat Rev Microbiol 2005 ; 3 : 711 – 721 . doi: 10.1038/nrmicro1234 OpenUrl CrossRef PubMed Web of Science 12. ↵ Parks DH et al. A standardized bacterial taxonomy based on genome phylogeny substantially revises the tree of life . Nat Biotechnol 2018 ; 36 : 996 – 1004 . doi: 10.1038/nbt.4229 OpenUrl CrossRef PubMed 13. ↵ Sánchez-Baracaldo P , Cardona T. On the origin of oxygenic photosynthesis and Cyanobacteria . New Phytol 2020 ; 225 : 1440 – 1446 . doi: 10.1111/nph.16249 OpenUrl CrossRef PubMed 14. ↵ Martiny AC , Treseder K , Pusch G. Phylogenetic conservatism of functional traits in microorganisms . ISME J 2013 ; 7 : 830 – 838 . doi: 10.1038/ismej.2012.160 OpenUrl CrossRef PubMed Web of Science 15. ↵ Ramoneda J et al. Leveraging genomic information to predict environmental preferences of bacteria . ISME J 2024 ; 18 : wrae195 . doi: 10.1093/ismejo/wrae195 OpenUrl CrossRef 16. Barberán A et al. Hiding in Plain Sight: Mining Bacterial Species Records for Phenotypic Trait Information . mSphere 2017 ; 2 : e00237 – 17 . doi: 10.1128/mSphere.00237-17 OpenUrl CrossRef PubMed 17. Madin JS et al. A synthesis of bacterial and archaeal phenotypic trait data . Sci Data 2020 ; 7 : 170 . doi: 10.1038/s41597-020-0497-4 OpenUrl CrossRef PubMed 18. ↵ Asplund-Samuelsson J , Hudson EP . Wide range of metabolic adaptations to the acquisition of the Calvin cycle revealed by comparison of microbial genomes . PLoS Comput Biol 2021 ; 17 : e1008742 . doi: 10.1371/journal.pcbi.1008742 OpenUrl CrossRef 19. Garritano AN , Song W , Thomas T. Carbon fixation pathways across the bacterial and archaeal tree of life . PNAS Nexus 2022 ; 1 : pgac226 . doi: 10.1093/pnasnexus/pgac226 OpenUrl CrossRef 20. ↵ Thiel V , Tank M , Bryant DA . Diversity of Chlorophototrophic Bacteria Revealed in the Omics Era . Annu Rev Plant Biol 2018 ; 69 : 21 – 49 . doi: 10.1146/annurev-arplant-042817-040500 OpenUrl CrossRef PubMed 21. ↵ Dos Santos PC et al. Distribution of nitrogen fixation and nitrogenase-like sequences amongst microbial genomes . BMC Genomics 2012 ; 13 : 162 . doi: 10.1186/1471-2164-13-162 OpenUrl CrossRef PubMed 22. ↵ Su Z , Wen D. Characterization of antibiotic resistance across Earth’s microbial genomes . Sci Total Environ 2022 ; 816 : 151613 . doi: 10.1016/j.scitotenv.2021.151613 OpenUrl CrossRef PubMed 23. ↵ Weissman JL , Hou S , Fuhrman JA . Estimating maximal microbial growth rates from cultures, metagenomes, and single cells via codon usage patterns . Proc Natl Acad Sci 2021 ; 118 : e2016810118 . doi: 10.1073/pnas.2016810118 OpenUrl Abstract / FREE Full Text 24. ↵ Cimen E , Jensen SE , Buckler ES . Building a tRNA thermometer to estimate microbial adaptation to temperature . Nucleic Acids Res 2020 ; 48 : 12004 – 12015 . doi: 10.1093/nar/gkaa1030 OpenUrl CrossRef PubMed 25. Li G et al. Machine Learning Applied to Predicting Microorganism Growth Temperatures and Enzyme Catalytic Optima . ACS Synth Biol 2019 ; 8 : 1411 – 1420 . doi: 10.1021/acssynbio.9b00099 OpenUrl CrossRef PubMed 26. ↵ Sauer DB , Wang D-N. Predicting the optimal growth temperatures of prokaryotes using only genome derived features . Bioinformatics 2019 ; 35 : 3224 – 3231 . doi: 10.1093/bioinformatics/btz059 OpenUrl CrossRef 27. ↵ Ramoneda J et al. Building a genome-based understanding of bacterial pH preferences . Sci Adv 2023 ; 9 : eadf8998 . doi: 10.1126/sciadv.adf8998 OpenUrl CrossRef PubMed 28. ↵ Jabłońska J , Tawfik DS . The number and type of oxygen-utilizing enzymes indicates aerobic vs. anaerobic phenotype . Free Radic Biol Med 2019 ; 140 : 84 – 92 . doi: 10.1016/j.freeradbiomed.2019.03.031 OpenUrl CrossRef PubMed 29. ↵ Gorman ED , Lladser ME . Interpretable metric learning in comparative metagenomics: The adaptive Haar-like distance . PLOS Comput Biol 2024 ; 20 : e1011543 . doi: 10.1371/journal.pcbi.1011543 OpenUrl CrossRef PubMed 30. ↵ Goberna M , Verdú M. Predicting microbial traits with phylogenies . ISME J 2016 ; 10 : 959 – 967 . doi: 10.1038/ismej.2015.171 OpenUrl CrossRef PubMed 31. ↵ Aslam S et al. Aerobic prokaryotes do not have higher GC contents than anaerobic prokaryotes, but obligate aerobic prokaryotes have . BMC Evol Biol 2019 ; 19 : 35 . doi: 10.1186/s12862-019-1365-8 OpenUrl CrossRef PubMed 32. Hildebrand F , Meyer A , Eyre-Walker A. Evidence of Selection upon Genomic GC-Content in Bacteria . PLOS Genet 2010 ; 6 : e1001107 . doi: 10.1371/journal.pgen.1001107 OpenUrl CrossRef PubMed 33. Hurst LD , Merchant AR . High guanine-cytosine content is not an adaptation to high temperature: a comparative analysis amongst prokaryotes . Proc R Soc B Biol Sci 2001 ; 268 : 493 – 497 . doi: 10.1098/rspb.2000.1397 OpenUrl CrossRef PubMed Web of Science 34. ↵ Sabath N et al. Growth Temperature and Genome Size in Bacteria Are Negatively Correlated, Suggesting Genomic Streamlining During Thermal Adaptation . Genome Biol Evol 2013 ; 5 : 966 – 977 . doi: 10.1093/gbe/evt050 OpenUrl CrossRef PubMed 35. ↵ Mcewan CEA , Gatherer D , Mcewan NR . Nitrogen-Fixing Aerobic Bacteria have Higher Genomic GC Content than Non-Fixing Species within the Same Genus . Hereditas 1998 ; 128 : 173 – 178 . doi: 10.1111/j.1601-5223.1998.00173.x OpenUrl CrossRef PubMed Web of Science 36. ↵ Brewer TE et al. Genome reduction in an abundant and ubiquitous soil bacterium ‘Candidatus Udaeobacter copiosus’ . Nat Microbiol 2016 ; 2 : 1 – 7 . doi: 10.1038/nmicrobiol.2016.198 OpenUrl CrossRef 37. ↵ Tian R et al. Small and mighty: adaptation of superphylum Patescibacteria to groundwater environment drives their genome simplicity . Microbiome 2020 ; 8 : 51 . doi: 10.1186/s40168-020-00825-w OpenUrl CrossRef PubMed 38. ↵ Williams TJ et al. Shedding Light on Microbial “Dark Matter”: Insights Into Novel Cloacimonadota and Omnitrophota From an Antarctic Lake . Front Microbiol 2021 ; 12 . 39. ↵ Belliveau NM et al. Fundamental limits on the rate of bacterial growth and their influence on proteomic composition . Cell Syst 2021 ; 12 : 924 - 944.e2 . doi: 10.1016/j.cels.2021.06.002 OpenUrl CrossRef PubMed 40. ↵ Dragone NB et al. Taxonomic and genomic attributes of oligotrophic soil bacteria . ISME Commun 2024 ; 4 : ycae081 . doi: 10.1093/ismeco/ycae081 OpenUrl CrossRef 41. ↵ Weissman JL et al. Benchmarking community-wide estimates of growth potential from metagenomes using codon usage statistics. 2022 . bioRxiv , 2022 ., 2022.04.12.488109 42. ↵ Berg IA . Ecological Aspects of the Distribution of Different Autotrophic CO2 Fixation Pathways . Appl Environ Microbiol 2011 ; 77 : 1925 – 1936 . doi: 10.1128/AEM.02473-10 OpenUrl Abstract / FREE Full Text 43. ↵ Lee JH et al. Expression and regulation of ribulose 1,5-bisphosphate carboxylase/oxygenase genes in Mycobacterium sp. strain JC1 DSM 3803 . J Microbiol 2009 ; 47 : 297 – 307 . doi: 10.1007/s12275-008-0210-3 OpenUrl CrossRef PubMed 44. ↵ Caldwell PE , MacLean MR , Norris PR . Ribulose bisphosphate carboxylase activity and a Calvin cycle gene cluster in Sulfobacillus species . Microbiology 2007 ; 153 : 2231 – 2240 . doi: 10.1099/mic.0.2007/006262-0 OpenUrl CrossRef PubMed Web of Science 45. Zakharchuk LM et al. Activity of the Enzymes of Carbon Metabolism in Sulfobacillus sibiricus under Various Conditions of Cultivation . Microbiology 2003 ; 72 : 553 – 557 . doi: 10.1023/A:1026039132408 OpenUrl CrossRef 46. Berg IA et al. Carbon Metabolism of Filamentous Anoxygenic Phototrophic Bacteria of the Family Oscillochloridaceae . Microbiology 2005 ; 74 : 258 – 264 . doi: 10.1007/s11021-005-0060-5 OpenUrl CrossRef Web of Science 47. ↵ Narsing Rao MP et al. Metagenomic analysis further extends the role of Chloroflexi in fundamental biogeochemical cycles . Environ Res 2022 ; 209 : 112888 . doi: 10.1016/j.envres.2022.112888 OpenUrl CrossRef 48. Shih PM , Ward LM , Fischer WW . Evolution of the 3-hydroxypropionate bicycle and recent transfer of anoxygenic photosynthesis into the Chloroflexi . Proc Natl Acad Sci 2017 ; 114 : 10749 – 10754 . doi: 10.1073/pnas.1710798114 OpenUrl Abstract / FREE Full Text 49. Ward LM , Shih PM . Phototrophy and carbon fixation in Chlorobi postdate the rise of oxygen . PLOS ONE 2022 ; 17 : e0270187 . doi: 10.1371/journal.pone.0270187 OpenUrl CrossRef PubMed 50. Fischer WW , Hemp J , Johnson JE . Evolution of Oxygenic Photosynthesis . Annu Rev Earth Planet Sci 2016 ; 44 : 647 – 683 . doi: 10.1146/annurev-earth-060313-054810 OpenUrl CrossRef 51. ↵ Khademian M , Imlay JA . How Microbes Evolved to Tolerate Oxygen . Trends Microbiol 2021 ; 29 : 428 – 440 . doi: 10.1016/j.tim.2020.10.001 OpenUrl CrossRef PubMed 52. ↵ Nielsen DA et al . Aerobic bacteria and archaea tend to have larger and more versatile genomes . Oikos 2021 ; 130 : 501 – 511 . doi: 10.1111/oik.07912 OpenUrl CrossRef 53. Nishida H. Evolution of genome base composition and genome size in bacteria . Front Microbiol 2012 ; 3 . 54. ↵ Rodríguez-Gijón A et al. A Genomic Perspective Across Earth’s Microbiomes Reveals That Genome Size in Archaea and Bacteria Is Linked to Ecosystem Type and Trophic Strategy . Front Microbiol 2022 ; 12 . 55. ↵ Miller J et al. ENS-t-SNE: Embedding Neighborhoods Simultaneously t-SNE . 2024 IEEE 17th Pac. Vis. Conf. PacificVis . 2024 . 2024 , pp 222 – 231 . OpenUrl 56. Green JL , Bohannan BJM , Whitaker RJ . Microbial Biogeography: From Taxonomy to Traits . Science 2008 ; 320 : 1039 – 1043 . doi: 10.1126/science.1153475 OpenUrl Abstract / FREE Full Text 57. Krause S et al. Trait-based approaches for understanding microbial biodiversity and ecosystem functioning . Front Microbiol 2014 ; 5 . doi: 10.3389/fmicb.2014.00251 OpenUrl CrossRef PubMed 58. Louca S et al. Function and functional redundancy in microbial systems . Nat Ecol Evol 2018 ; 2 : 936 – 943 . doi: 10.1038/s41559-018-0519-1 OpenUrl CrossRef PubMed 59. Weiher E , Keddy PA . Assembly Rules, Null Models, and Trait Dispersion: New Questions from Old Patterns . Oikos 1995 ; 74 : 159 – 164 . doi: 10.2307/3545686 OpenUrl CrossRef Web of Science 60. ↵ McGill BJ et al. Rebuilding community ecology from functional traits . Trends Ecol Evol 2006 ; 21 : 178 – 185 . doi: 10.1016/j.tree.2006.02.002 OpenUrl CrossRef PubMed Web of Science 61. ↵ Matchado MS et al. On the limits of 16S rRNA gene-based metagenome prediction and functional profiling . Microb Genomics 2024 ; 10 : 001203 . doi: 10.1099/mgen.0.001203 OpenUrl CrossRef 62. ↵ Parks DH et al. CheckM: assessing the quality of microbial genomes recovered from isolates, single cells, and metagenomes . Genome Res 2015 ; 25 : 1043 – 1055 . doi: 10.1101/gr.186072.114 OpenUrl Abstract / FREE Full Text 63. ↵ Hyatt D et al. Prodigal: prokaryotic gene recognition and translation initiation site identification . BMC Bioinformatics 2010 ; 11 : 119 . doi: 10.1186/1471-2105-11-119 OpenUrl CrossRef PubMed 64. ↵ Kramer O Kramer O. Scikit-Learn . In: Kramer O (ed.), Machine Learning for Evolution Strategies . Cham : Springer International Publishing , 2016 , 45 – 53 . 65. ↵ Zimmermann J , Kaleta C , Waschina S. gapseq: informed prediction of bacterial metabolic pathways and reconstruction of accurate metabolic models . Genome Biol 2021 ; 22 : 81 . doi: 10.1186/s13059-021-02295-1 OpenUrl CrossRef PubMed 66. ↵ Camacho C et al. BLAST+: architecture and applications . BMC Bioinformatics 2009 ; 10 : 421 . doi: 10.1186/1471-2105-10-421 OpenUrl CrossRef PubMed 67. ↵ Waskom ML . seaborn: statistical data visualization . J Open Source Softw 2021 ; 6 : 3021 . doi: 10.21105/joss.03021 OpenUrl CrossRef 68. ↵ Harris CR et al. Array programming with NumPy . Nature 2020 ; 585 : 357 – 362 . doi: 10.1038/s41586-020-2649-2 OpenUrl CrossRef PubMed 69. ↵ Hunter JD . Matplotlib: A 2D Graphics Environment . Comput Sci Eng 2007 ; 9 : 90 – 95 . doi: 10.1109/MCSE.2007.55 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted July 17, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following A periodic table of bacteria?: Mapping bacterial diversity in trait space Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share A periodic table of bacteria?: Mapping bacterial diversity in trait space Michael Hoffert , Evan Gorman , Manuel E. Lladser , Noah Fierer bioRxiv 2025.07.11.664459; doi: https://doi.org/10.1101/2025.07.11.664459 Share This Article: Copy Citation Tools A periodic table of bacteria?: Mapping bacterial diversity in trait space Michael Hoffert , Evan Gorman , Manuel E. Lladser , Noah Fierer bioRxiv 2025.07.11.664459; doi: https://doi.org/10.1101/2025.07.11.664459 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7617) Biochemistry (17633) Bioengineering (13856) Bioinformatics (41841) Biophysics (21399) Cancer Biology (18529) Cell Biology (25422) Clinical Trials (138) Developmental Biology (13352) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24281) Genetics (15582) Genomics (22461) Immunology (17700) Microbiology (40295) Molecular Biology (17140) Neuroscience (88413) Paleontology (666) Pathology (2823) Pharmacology and Toxicology (4813) Physiology (7632) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4284) Systems Biology (9808) Zoology (2267)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.