Full text
112,211 characters
· extracted from
preprint-html
· click to expand
Tracing the vertebrate selenoproteome evolution reveals expansions in ray-finned fishes and convergent depletions in tetrapods | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Tracing the vertebrate selenoproteome evolution reveals expansions in ray-finned fishes and convergent depletions in tetrapods Max Ticó , View ORCID Profile Jesus Lozano-Fernandez , View ORCID Profile Marco Mariotti doi: https://doi.org/10.1101/2025.05.29.656587 Max Ticó 1 Department of Genetics, Microbiology and Statistics, Universitat de Barcelona , Barcelona, Catalonia, 08024, Spain Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jesus Lozano-Fernandez 1 Department of Genetics, Microbiology and Statistics, Universitat de Barcelona , Barcelona, Catalonia, 08024, Spain Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jesus Lozano-Fernandez Marco Mariotti 1 Department of Genetics, Microbiology and Statistics, Universitat de Barcelona , Barcelona, Catalonia, 08024, Spain Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Marco Mariotti For correspondence: marco.mariotti{at}ub.edu Abstract Full Text Info/History Metrics Supplementary material Preview PDF ABSTRACT Selenoproteins incorporate the rare selenium-containing amino acid selenocysteine (Sec) and play crucial roles for redox homeostasis, stress response, and hormone regulation. Sec is inserted by co-translational recoding of the UGA codon, normally a stop. As a consequence, selenoproteins are often misannotated in public databases and require specialized bioinformatic methods and resources. Here, we present a refined characterization of the composition and evolution of the vertebrate selenoproteome. Based on analyses of 19 gene families across hundreds of genomes, we show that extant selenoproteomes were shaped by extensive gene duplications (56 selenoproteins), losses (50), and Sec-to-cysteine (Cys) conversions (21). Tetrapods including mammals encode 24-25 selenoproteins, with variations in 6 families. Notably, the same genes underwent convergent evolutionary events in multiple tetrapods, namely Sec-to-Cys substitutions ( SELENOU1 , GPX6 ) and gene losses ( SELENOV ). In contrast, ray-finned fish exhibit larger and more dynamic selenoproteomes, reinforcing the hypothesis that the selective advantage of Sec is stronger in aquatic environments. We detected selenoprotein duplications spread across the actinopterygian clade involving 13 families, mainly involved in antioxidant defense. The richest selenoproteomes were found in Salmonidae and Cyprinoidei fish with 56 and 44 selenoproteins, respectively, owing to whole genome duplications. Among our findings, the SELENOP family stands out in lampreys, carrying up to an unprecedented 162 UGAs putatively recoded to Sec. Our study presents the most comprehensive evolutionary map of vertebrate selenoproteins to date and delineates the specific selenoproteome of each lineage, establishing a foundational framework for selenium biology research in the era of biodiversity genomics. BACKGROUND Selenium is an essential trace element for human health. The biological effects of selenium are mostly mediated by its incorporation in the non-canonical amino acid selenocysteine (Sec) [ 1 – 3 ]. Sec is the essential component of selenoproteins, a unique class of proteins with crucial functions in both normal physiology and cancer. Sec is often referred to as the 21st amino acid due to its distinctive biosynthetic pathway and incorporation mechanism, which differ fundamentally from those of the 20 canonical amino acids. Importantly, Sec is inserted in correspondence to the UGA codon, which is normally a stop codon. The UGA is co-translationally recoded to Sec in selenoprotein mRNAs in response to local sequence elements, the major of which is called the SECIS (selenocysteine insertion sequence) [ 4 , 5 ]. Moreover, unlike standard amino acids which are synthesized then charged onto their cognate transfer RNAs (tRNAs), Sec is synthesized directly on its own dedicated tRNA, tRNASec, through a multi-step enzymatic process. The selenium required for this reaction is provided in the form of selenophosphate, synthesized by protein SEPHS2, an essential enzyme in Sec metabolism which, intriguingly, is itself a selenoprotein [ 6 ]. Since their discovery in the 1970s, Sec and selenoproteins have been extensively studied across diverse fields of biology [ 7 , 8 ]. Indeed, despite representing a small fraction of protein-coding genes, selenoproteins perform many critical functions in redox homeostasis, stress response, hormone maturation, and protein quality control, thus playing key roles in both physiology and cancer biology. This has prompted extensive research into their enzymatic functions, selenium metabolism, and the unique mechanism of Sec biosynthesis. Selenium itself, a trace element with a narrow margin between deficiency and toxicity, has also drawn interest in nutritional and physiological contexts [ 3 ]. As a result, selenoproteins are investigated in a broad range of species, from standard model organisms (e.g., Mus musculus, Danio rerio ) to agriculturally relevant animals (e.g., Bos taurus, Gallus gallus, Salmo salar ). Naturally, understanding the full set of selenoproteins encoded by an organism is essential for the meaningful design and interpretation of experiments to elucidate selenium biology. The human and murine selenoproteomes were first described in 2003 as comprising 25 and 24 Sec-encoding genes, respectively [ 9 ]. Subsequent studies expanded the known selenoproteomes of selected vertebrates and metazoans [ 10 – 15 ]. In our seminal paper in 2012, we provided the first description of selenoproteome composition and evolution across diverse vertebrates [ 16 ]. Although highly influential, that study was constrained by the limited availability of sequenced genomes at the time (44). In the last 10 years, major advances in sequencing and the rise of biodiversity genomics provided an unprecedented wealth of genomic data [ 17 , 18 ]. However, due to their deviation from the standard genetic code, selenoproteins have not kept pace with the expansion of automated genomics resources. At present, ∼90% and ∼50% of selenoprotein genes have substantial annotation errors in Ensembl and NCBI genomes, respectively [ 19 ]. This also led to unreliable selenoprotein data in evolutionary resources dedicated to orthology (e.g. Ensembl Compara [ 20 ], EggNOG [ 21 ]). In this manuscript, we present a refined comprehensive characterization of the selenoproteome of vertebrates, obtained by meticulously tracing the evolution of 19 protein families across hundreds of genomes through sequence analysis and phylogenetic reconstruction. Our work provides an essential evolutionary framework to understand the selenoprotein content of any vertebrate genome, expanding on previous efforts by an order of magnitude. RESULTS We downloaded 314 genome assemblies from Ensembl v.108 representative of all major lineages of vertebrates, complemented by 26 NCBI genomes from early vertebrates and outgroups (Methods). Next, we used the software Selenoprofiles (Methods) to identify homologs belonging to any of the known metazoan selenoprotein families in those genomes, resulting in 13,517 gene predictions. Phylogenetic reconstruction was then performed for each family, and gene and species trees were manually inspected to infer gene evolutionary events (Methods). These comprise gene duplications, gene losses, and the selenoprotein-specific event of the conversion of a Sec residue to Cys. In selenoprotein evolution, Cys to Sec events are also possible, but have been observed only along evolutionary distances much larger than encompassed by vertebrates [ 22 , 23 ]. Each branch of gene trees was manually annotated as a separate orthologous group, assigning gene names as consistent as possible with existing resources and literature. Tree inspection and orthology assignments were performed taking into account branch support values obtained by UltraFast Bootstrap (UFB). The great majority of orthologous groups were supported by high UFB support values in gene trees (Methods); all exceptions are explicitly mentioned in their relevant section. Each putative event was critically assessed, following criteria of maximum parsimony, and seeking consistency with the species tree and support by independent data (e.g. RNA sequences, other genome assemblies from NCBI) (Methods). Thanks to the vast diversity of organisms in our dataset, the great majority of events were shared among multiple species, leading to phylogenetically coherent patterns of gene occurrence that allowed reliable evolutionary inferences. Yet, two lineages stood out for their divergence from this general trend. First, bird genomes exhibited many apparent gene losses scattered across lineages, in a pattern that would imply dozens of independent events (Supplementary Figure S1). This included the apparent absence of SEPHS2 , a gene that nevertheless is required for biosynthesis of all other selenoproteins [ 22 ], and of SELENOW . We ascribed this pattern to a known bias of bird assemblies: they systematically lack certain GC-rich regions that present technical challenges for sequencing [ 24 ]. To further support this interpretation, we examined the syntenic block surrounding SEPHS2 , which includes the essential genes G6PD and IKBKG (Figure S2). This genomic region is conserved across Sauria; yet, we observed the apparent absence of all three genes in most (but not all) Aves. In contrast, outgroup lineages such as Testudines retain the complete syntenic block, including the three genes. Strikingly, we found RNA evidence of expression for both genes ( SEPHS2 and SELENOW ) in multiple species despite their apparent absence from their available genome assemblies, demonstrating that these genes are present in the actual genomes of birds. Ultimately, we dismissed all selenoprotein losses in Sauria (which includes birds) as artifactual. Second, the fish lineage of Cyprinoidei (e.g. carps, Danio rerio ) exhibited a massive amount of apparent gene duplications (Supplementary Figure S1), which we could not precisely disentangle by phylogenetic analysis. We realized that the sublineage of Cyprininae (e.g. carps) reportedly underwent extensive recent whole genome duplications and polyploidy, which greatly complicated this task. Therefore, we decided to limit our analysis to the evolutionary events that occurred at the root of Cyprinoidei, i.e. shared between D. rerio and Cyprininae, and dismiss the rest. Several of the outgroups we used have a characterized selenoproteome: Ciona intestinalis [ 15 ], Branchiostoma floridae [ 14 ], Drosophila melanogaster [ 25 ], Strigamia maritima [ 26 , 27 ]. This allowed us to critically assess their gene predictions, and compare them with those of vertebrates to phase evolutionary events, as detailed later for each family. Overview of the vertebrate selenoproteome We identified selenoproteins belonging to 19 protein families. The most represented families were glutathione peroxidases with 1,529 Sec-encoding genes, iodothyronine deiodinases with 988, and thioredoxin reductases with 718. In general, Actinopterygii (ray-finned fishes) exhibited richer selenoproteomes (35-56 selenoprotein genes per genome) than Tetrapoda (24-25 genes per genome) ( Figure 1 ). Download figure Open in new tab Figure 1. Evolution of the vertebrate selenoproteome. The evolutionary events shaping selenoprotein genes are shown along the species tree. Leaf nodes represent either individual species (italicized) or collapsed lineages; in the latter case, the number in brackets indicates how many species from that lineage are included in our main dataset. Asterisks (‘*’) denote lineages with apparent gene losses that we interpret as artifacts caused by known issues in genome assembly quality. Bold numbers in the rightmost column indicate the number of selenoprotein genes detected in each species or lineage. Grey rectangles highlight taxonomic clades of interest; “Cyclost.” Stands for Cyclostomata, and “Protacant” for Protacanthopterygii. Red dots mark whole genome duplications (WGD) reported in literature [ 74 – 79 ]. Animal silhouettes were sourced from https://www.phylopic.org . While our analyses recapitulated earlier reports performed at a smaller scale, we also detected many novel events, particularly gene duplications. The majority affected Actinopterygii and its sublineages, whose selenoproteome appeared not only larger, but also more dynamic. In total, we detected 56 gene duplications (37 in Actinopterygii), and 50 gene losses (29 in Actinopterygii). The most selenoprotein-rich organisms were Salmonidae (aka salmonids, e.g. trouts, salmons) with 56 selenoprotein genes, 17 of which emerged by duplication specifically in this lineage, and Cyprinoidei with 44 selenoproteins, 6 of which lineage-specific. On the other hand, Sec-to-Cys conversions were more prevalent in Tetrapoda: we detected 21 of these events overall, 16 of which were in Tetrapoda. Hereafter, we describe the selenoproteome evolution for each family separately. Glutathione peroxidases (GPX) Glutathione peroxidases (GPX) are key antioxidant enzymes that protect cells from oxidative damage by reducing hydrogen peroxide and lipid peroxides using glutathione as an electron donor [ 16 , 28 ]. Mammals possess eight GPX homologs with specialized tissue distribution and functions, including five selenoproteins: GPX1 (ubiquitous), GPX2 (gastrointestinal), GPX3 (extracellular, kidney), GPX4 (membrane-associated, ubiquitous), GPX6 (olfactory system); and three Cys paralogs: GPX5 , GPX7 , and GPX8 . Our phylogenetic analysis ( Figures 2 and 3 , Supplementary Figure S3) confirms a well-supported evolutionary classification of GPX proteins, identifying three major groups: GPX1/GPX2, GPX3/GPX5/GPX6, and GPX4/GPX7/GPX8. These findings align with previous studies on GPX evolution [ 16 , 28 , 29 ]. Download figure Open in new tab Figure 2. Phylogeny of the GPX family. The GPX gene tree was computed collating all sequences of this family, but it is split in different representations for visualization purposes. a) Topology of the GPX family, split into three clades shown in panels b, Figure 3a and 3b . b) Phylogeny of GPX3/5/6 clade. To the right of the main tree, a summarized tree representation is shown, displaying the evolutionary events identified in the full tree, with selected clades collapsed for visualization. Dotted lines in the summary tree represent paraphyletic, non-collapsible branches. Duplication events are indicated by blue nodes labeled with their UFB support values. The legend below shows the color scheme used to represent the taxonomic groups of the species highlighted in the tree. Note that certain monophyletic clades of the gene tree were collapsed (e.g. see GPX5). The full expanded (non-collapsed) gene tree is available as Supplementary Figure S3. c) Zoomed section of the GPX6 gene tree, used here to illustrate this form of representation, abundantly used throughout this paper. Each leaf corresponds to a putative gene whose protein sequence was included in the tree. Leaves are color-coded by their label, assigned by Selenoprofiles by examining which codon aligns to the Sec position in the profile (e.g. blue for Sec-UGA, orange for Cys, pink for unaligned). Moving outward, you will see an equally colored representation of the corresponding aligned protein, which shows its non-gap positions and includes black lines to map where introns are located in the coding sequence. Most intronless genes are pseudogenes, which constitute the prevalent false positives of Selenoprofiles. The outmost circle provides a manual annotation of gene clades, with colors that indicate the taxonomy of the source of those genes (e.g. blue for Actinopterygii, green for Mammalia). Internal nodes of the gene tree represent either speciation events (red) or duplication events (blue), determined via the species overlap algorithm. The latter are displayed with size proportional to the number of species supporting the duplication. The nodes that define duplication events deemed genuine via manual analysis are marked with a star. Branch lengths represent the expected number of substitutions per site and were estimated by maximum likelihood under the best-fitting substitution model in IQ-TREE. Download figure Open in new tab Figure 3. Phylogeny of the GPX family, continued from figure 2 . a, b) Gene tree visualization of the GPX1/2 and GPX4/7/8 clades, respectively. See full feature description in Figure 2c The full gene tree is available as Supplementary Figure S3. f) Multiple sequence alignment of GPX6 proteins including representatives that underwent Cys conversions. Sec residues are shown in blue, while Cys residues at this position are in orange. The identifiers of GPX6-Cys homologs are highlighted. GPX1, GPX2, GPX3, GPX4 are present in all vertebrates, with a single notable exception. In the GPX1/GPX2 group, we observed that the Hyperoartia class of jawless fish (Cyclostomata superclass) possessed two similar genes which phylogenetic reconstruction placed as outgroups of the GPX1 and GPX2 clusters of other vertebrates ( Figure 3a ). We then noticed that the other class of Cyclostomata, Myxini, has instead a single orthologous gene (Supplementary Figure S1). We infer that GPX1 and GPX2 split in Gnathostomata (jawed vertebrates) and that Cyclostomata likely reflect the ancestral status, so that we named this gene GPX1/2 and its Hyperoartia-specific paralog GPX1/2b. Although the UFB support of the corresponding node in the reconstructed protein tree is only 25%, we argue that we can confidently map the origin of GPX1/2b to Hyperoartia based on the species-tree analysis. GPX6 originated at the root of placentals and exhibited a very dynamic evolution. Multiple independent GPX6 Sec-to-Cys conversions had been reported [ 16 ], which we confirmed and expanded in our analyses ( Figures 1 , 2b , 2c , 3c ; Supplementary Figure S3). We observed Cys conversions in multiple taxonomic groups, including Platyrrhini primates (New World monkeys, e.g. Callithrix jacchus , Cebus imitator ), Muroidea rodents (e.g. Mus musculus , Rattus norvegicus , Mesocricetus auratus ), Lagomorpha (e.g. Oryctolagus cuniculus , Ochotona princeps ), Feliformia (e.g. Felis catus , Panthera pardus ), Bathyergidae rodents (e.g. Fukomys damarensis ). In addition, we also detected species-specific conversions, namely in Marmota marmota marmota, Sciurus vulgaris, Zalophus californianus and Castor canadensis ( Figure 3c ). Furthermore, GPX6 was lost altogether in Chiroptera (bats). We identified several GPX gene duplications across vertebrates. We confirmed two previously reported GPX gene duplications at the root of Actinopterygii leading to the emergence of GPX4b and GPX1b (the latter with only 20% UFB, but consistent occurrences along the species tree). These genes were found in all investigated species within this lineage, except for Oryziinae fish (e.g. Oryzias latipes ), which we infer to have lost GPX1b . In two Oryziinae species, we initially identified a putative Cys-containing GPX4b paralog, yet we dismissed it due to lack of sequence support from RNA data. A third gene, GPX3b , was also reported in Actinopterygii; our analysis indicates that it originated at the root of Vertebrata, and was subsequently converted to Cys in Myxini and lost in Tetrapoda. We also observed additional selenoprotein gene duplications in specific branches of Actinopterygii. In Hyperoartia, we observed three lineage-specific events that yielded GPX1/2b , GPX3b2 and GPX4c . Then, a lineage-specific duplication of GPX1 gave rise to a distinct Sec-paralog, GPX1c , in Chondrichthyes (cartilaginous fish). We mapped three additional duplications in Salmonidae yielding GPX1b2 , GPX3a2 , and GPX4a2 . All these events were supported by gene tree nodes with high UFB support values (>70%), except the aforementioned GPX1/2b. Notably, these duplications were previously reported in an analysis of 8 fish genomes [ 29 ]. Our study demonstrates that they are lineage-specific events, so that we have assigned new names to the duplicated genes to reflect their precise evolutionary trajectories. The correspondence with [ 29 ] is provided in Supplementary Table T1. Iodothyronine deiodinases (DIO) Iodothyronine deiodinases (DIO) regulate thyroid hormone metabolism by catalyzing the activation or inactivation of thyroxine (T4) and triiodothyronine (T3) [ 30 ]. The vertebrate DIO family includes three main genes: DIO1 , which is involved in peripheral thyroid hormone conversion; DIO2 , which plays a critical role in local T3 activation; and DIO3 , which inactivates excess thyroid hormones, preventing hyperthyroidism. Through our analyses ( Figures 1 , 4 , Supplementary Figures S1, S4), we detected these three genes in all vertebrate lineages, with the exception of Cyclostomata that lacked DIO1 . This gene is present in all investigated Chordates, indicating that DIO1 predates the vertebrate radiation and was then lost in Cyclostomata. All Actinopterygii additionally carried a DIO3 duplicate ( DIO3b ). The placement of DIO3b genes in the reconstructed protein tree was mostly consistent with an origin by DIO3 duplication at the root of Actinopterygii (UFB support of relevant node: 68%), with the exception of Cyprinoidei DIO3b which was placed closer to the tree root, possibly due to long branch attraction. Download figure Open in new tab Figure 4. Phylogeny of the DIO family. a) Gene tree visualization of the DIO family. See full feature description in Figure 2c . Expanded gene and species trees are available as Supplementary Figures S4 and S5. We also confirmed two more duplications, one of the ancestral DIO3 gene (also called DIO3a ) [ 31 ], to generate DIO3a2 , and the other of DIO2 to yield DIO2b ( Figures 1 , 4 , Supplementary Figures S4, S5). Both were mapped to Salmonidae with high support, as previously reported [ 31 ]. In Hyperoartia (Cyclostomata), we identified two lineage-specific duplications resulting in DIO2c and DIO3c (absent from Myxini, the other class of Cyclostomata; Supplementary Figure S1). Both were highly supported, with the caveat that DIO2c was placed as an outgroup of all DIO2 genes, rather than at its expected position as sister of Cyclostomata DIO2 . Thioredoxin reductases (TXNRD) Thioredoxin reductases (TXNRDs) play a central role in maintaining cellular redox balance by catalyzing the reduction of thioredoxins, which are essential for regulating oxidative stress responses, cell proliferation, and redox-regulated signaling cascades [ 32 ]. Tetrapods have three TXNRD genes: TXNRD1 (cytosolic), TXNRD2 (mitochondrial), and TXNRD3 (also known as TGR ; testis-specific, containing a peculiar N-terminal Grx domain), each of which participates in distinct physiological processes. In phylogenetic reconstruction, TXNRD1 and TXNRD3 clearly cluster together, indicating a shared origin. Reportedly, fishes lack the TXNRD3/TGR gene, but their TXNRD1 gene encodes for a Grx-containing protein as its major isoform, which duplicated to give rise to TXNRD3/TGR in tetrapods [ 16 , 33 ]. Yet, our present analysis ( Figure 5 , Supplementary Figures S6-S9) did not recapitulate TXNRD evolution as previously proposed. While indeed nearly all fish possess a single gene in the TXNRD1/TXNRD3 group, we claim here that it is the ortholog of TXNRD3 , rather than TXNRD1 . This classification is supported by both the topology of reconstructed gene trees ( Figure 5a , Supplementary Figure S6) and by syntenic conservation (Supplementary Figures S7, S8). Furthermore, we detected both genes ( TXNRD1 and TXNRD3) in Chondrichthyes. Because this lineage serves as an outgroup to both Actinopterygii and Tetrapoda, we argue that TXNRD1 emerged earlier than tetrapods. To explain its sparse occurrence across Actinopterygii, we hypothesize that it was subsequently lost independently in six Actinopterygian lineages ( Figure 1 ). Curiously, fish TXNRD1 features an atypical GNCUG C-terminus (rather than GCUC found across TXNRDs). We did not find TXNRD1 in Cyclostomata, and thus we map the TXNRD1/TXNRD3 duplication to the root of Gnathostomata. Finally, because the Cyclostomata gene clearly clusters with TXNRD3 , we identify the TXNRD3 clade as the ancestral one and TXNRD1 as the gene derived from its duplication. Download figure Open in new tab Figure 5. Phylogeny of the TXNRD family. a) Gene tree visualization of the TXNRD family. See full feature description in Figure 2c . Expanded gene and species trees are available as Supplementary Figures S6, while Supplementary Figures S7 and S8 contain analyses of the synteny of TXNRD1 and TXNRD3 genes. b) Phylogenetic distribution of the Grx domain in TXNRD1 and TXNRD3 proteins across vertebrate lineages. Colors are assigned according to the TXNRD gene tree. A non-collapsed version of this figure is available as Supplementary Figure S9. It is reported that the human TXNRD1 gene encodes for a minor transcript isoform that includes the Grx domain (“TxnRd1_v3”), which is a predominant feature of TXNRD3 genes. This transcript isoform is reportedly absent in M. musculus TXNRD1 [ 32 ]. Therefore, we systematically investigated the occurrence of Grx domains in the genomic regions of TXNRD1 and TXNRD3 across all vertebrates (Methods). Our results show that Grx domains are a widespread feature of TXNRD1 genes ( Figure 5b , Supplementary Figure S9). However, they have been lost in several tetrapod lineages independently (Rodentia, Aves, Equus, Prototheria, Metatheria and Pecora) and in all Actinopterygii retaining the TXNRD1 gene, possibly due to functional redundancy with TXNRD3 . Altogether, we argue that the Grx-domain is an ancestral feature of TXNRD3 which was initially retained in TXNRD1 , but that was later lost in multiple lineages through convergent evolution. We further investigated the ancestral state of vertebrates by examining the position of other metazoan groups in the TXNRD gene tree ( Figure 5a ), as well as an extended gene tree featuring additional outgroups available in PhylomeDB [ 34 ] (QFO2020 seed, Caenorhabditis elegans phylome ID: 791). Our phylogenetic analyses indicate that the divergence between TXNRD1/TXNRD3 and TXNRD2 likely predates the origin of metazoans, as representatives from both clades are found in diverse animal lineages, including the nematode C. elegans , the annelid Helobdella robusta , and the cnidarian Nematostella vectensis . Secondary events are evident in several lineages, and detailed hereafter. The tunicate Ciona intestinalis lost TXNRD2 : we only detected one TXNRD homolog in this species, consistent with an earlier report [ 15 ], which clusters with TXNRD1/TXNRD3 in the reconstructed gene tree (Supplementary Figure S6). Analogous observations indicate that arthropods such as Ixodes scapularis and D. melanogaster lost TXNRD1 (Supplementary Figure S6). Notably, D. melanogaster retains two TXNRD2 -derived genes, presumably the result of a secondary duplication. Fitting with our analysis, non-vertebrate animals carried Grx-domains solely in TXNRD3 genes. Due to the presence of this domain, these invertebrate genes were referred as TGR in literature, despite weak or absent glutathione reductase activity [ 35 ]. Altogether, we reconstruct that ancestral vertebrates carried a cytosolic Grx-containing TXNRD3 gene and a mitochondrial TXNRD2 gene, the former later giving rise to TXNRD1 in Gnathostomata. We further detected one TXNRD3 duplication occurred in Salmonidae and yielded TXNRD3b , which retained the Grx domain. Lastly, the Hyperoartia clade of Cyclostomata lost TXNRD2 (despite retaining its neighboring genes) and thus encodes TXNRD3 as their sole gene in this family. TXNRD2 was detected in its sister clade, Myxini (Supplementary Figure S1). SELENOJ Selenoprotein J (SELENOJ) shows significant similarity to jellyfish J1-crystallins and, together with them, forms a distinct subfamily within the large family of ADP-ribosylation enzymes. While the homology may suggest a structural role which would be unique among selenoproteins, SELENOJ remains functionally uncharacterized. The SELENOJ gene is reportedly restricted to actinopterygian fishes and sea urchins, with Cys homologs reported in cnidarians [ 11 ]. Our analyses replicated these earlier findings, and additionally highlighted the presence of SELENOJ in Amphioxiformes (Supplementary Figure S10). On the other hand, SELENOJ was lost in Cyclostomata and Batoidea (rays e.g Leucoraja erinaceus )(Supplementary Figure S1). Moreover, we had previously reported a SELENOJ duplication in O. latipes and Gasterosteus aculeatus [ 16 ], yielding SELENOJ2 . Here, we traced back this duplication to the root of Clupeocephala (76% UFB support), which included more than half of fishes in our dataset ( Figures 1 , 6a , Supplementary Figures S10, S11). Later, SELENOJ2 was lost independently in Otophysi (e.g Danio rerio) and Tetraodontidae (e.g. Tetraodon nigroviridis ). On the other hand, SELENOJ underwent a Sec-to-Cys conversion in Salmonidae. Download figure Open in new tab Figure 6. Phylogeny of the SELENOJ, SELENOO, and MSRB families. a, b, c) Gene tree visualizations of the SELENOJ, SELENOO, and MSRB families, respectively. See full feature description in Figure 2c . Expanded gene and species trees are available: Supplementary Figure S10, S11 (SELENOJ), S12, S13 (SELENOO), S14, S15 (MSRB). SELENOO Selenoprotein O (SELENOO) is the largest known selenoprotein by molecular weight, and belongs to a widespread protein family recently functionally characterized as a mitochondrial pseudokinase that transfers AMP to various protein substrates, implicated in oxidative stress response and metastasis [ 36 , 37 ]. Previous studies identified a SELENOO paralog named SELENOO2 in D. rerio , which was absent in the genomes of Takifugu rubripes , G. aculeatus , O. latipes , and T. nigroviridis [ 16 ]. In our analysis, we replicated these results, and yet we observed the presence of SELENOO2 in many other fish genomes, including among others all of Cyprinodontoidei (toothcarps) and Carangaria (e.g. jacks, flatfishes), which implies a complex pattern of gene evolution ( Figures 1 , 6b , Supplementary Figures S12, S13). Ultimately, we hypothesize that SELENOO2 emerged at the root of Clupeocephala but was subsequently lost in specific lineages, namely Tetraodontidae, Oryziinae, Anabantaria, and almost all Perciformes (e.g. G. aculeatus ). Methionine sulfoxide reductase B (MSRB) Methionine sulfoxide reductase B (MSRB, previously known as SELENOR), are enzymes that reduce oxidized methionine residues, with roles for antioxidant defense and redox-based modulation of various pathways including actin assembly [ 38 ]. Humans possess one Sec-containing gene, MSRB1 , and two Cys paralogs, MSRB2 and MSRB3 . Our analyses ( Figure 6c , Supplementary Figures S14 and S15) indicate that the three genes predate the origin of vertebrates, since members of each subfamily were detected in invertebrates. Within fish lineages, MSRB1 underwent a complex history of duplications and losses. We previously identified a Sec-containing gene, MSRB1b , in five actinopterygian species [ 16 ]. Based on the protein tree topology ( Figure 6c , Supplementary Figure S14; UFB support 58%) and the pattern of gene occurrence across the species tree (Supplementary Figure S15), we mapped its origin to the subclade of Osteoglossocephalai, which comprises all Actinopterygii in our dataset except the MSRB1b -lacking species Erpetoichthys calabaricus and Lepisosteus oculatus . Gene counts per species further indicated secondary MSRB1b duplications; yet, the reconstructed protein tree exhibited a scattered topology, without the typical expected clustering pattern ( Figure 6c ). Based mostly on the species tree, we concluded that MSRB1b likely duplicated in Salmonidae yielding MSRB1b2, and in Euacanthomorphacea yielding MSRB1b3; and that both MSRB1b and the duplicated copy MSRB1b3 have been later lost in multiple lineages: MSRB1b in Esociformes, Tetraodontidae and Anabantaria and MSRB1b3 in Perciformes, Cynoglossus semilaevis and Cyprinodontiformes. Notably, this reconstruction highlights at least three convergent events of MSRB1b3 loss in Percomorphacae, and another three loss events of MSRB1b in Euteleosteomorpha. Relevantly, other selenoprotein families show similar patterns of convergent losses in this lineage ( Figure 1 ), as discussed later. SELENOT Selenoprotein T (SELENOT) is a thioredoxin-like enzyme anchored to the endoplasmic reticulum (ER) membrane. It is highly conserved throughout evolution, and it has an essential role in cellular homeostasis and redox regulation [ 39 ]. While mammals have a single SELENOT gene, previous studies reported three homologs in actinopterygian fishes [ 16 ], here refined and expanded. We found that most actinopterygian groups possessed two SELENOT genes, namely SELENOT (aka SELENOT1 ) and SELENOT2 ; yet, Salmonidae, Otophysi and Cyprinoidei exhibited further lineage-specific duplications. Ultimately, we mapped the origin of SELENOT2 to a duplication of SELENOT1 at the root of Actinopterygii ( Figure 7a ). SELENOT1 duplicated again to yield SELENOT1b in Otophysi, i.e. the group including Cyprinoidei (e.g . D. rerio and carps), and Characiphysae (e.g. Electrophorus electricus, Pygocentrus nattereri). In Cyprinoidei, further duplications of SELENOT2 and SELENOT1b gave rise to the genes SELENOT2b and SELENOT1b2 , respectively ( Figures 1 , 7a , Supplementary Figure S16). Finally, we observed yet another SELENOT1 duplication in Salmonidae, yielding the gene SELENOT1a2 . Interestingly, Hyperoartia also independently expanded this family through one gene duplication, which yielded a divergent paralog that we designated SELENOT3 . We deem all SELENOT duplications highly reliable due to the consistency between the species tree and the protein tree topology, and to the high UFB values in the latter ( Figure 7a ). Download figure Open in new tab Figure 7. Phylogeny of the SELENOT, SELENOU families. a, b) Gene tree visualizations of the SELENOT and SELENOOU families, respectively. See full feature description in Figure 2c . Expanded gene trees are available as Supplementary Figures S16 (SELENOT) and S17 (SELENOU). c) SELENOU multiple sequence alignment. Species are color-labelled according to lineage. Sec residues are marked in blue and Cys residues in orange. SELENOU The Selenoprotein U (SELENOU) family includes selenoprotein genes in birds and fishes, the latter reported to carry up to three Sec-containing paralogs [ 10 , 16 ]. Mammals and amphibians are reported to encode only Cys-homologs, with human containing three paralogs ( SELENOU1/PRXL2A , SELENOU2/PRXL2C , SELENOU3/PRXL2B ), involved in diverse functions such as antioxidant defense, prostaglandin biosynthesis, and glycolysis regulation. The SELENOU gene family exhibits a scattered pattern of occurrence, implying a complex evolutionary history involving gene duplications, losses, and Sec-to-Cys conversions. Here, we provide our best attempt to explain their history with current data ( Figures 1 , 7b , 7c , Supplementary Figures S1, S17). The split of SELENOU1/2/3 predates vertebrates, as members of the three subfamilies were detected in invertebrate species (Supplementary Figures S17). All Sec-encoding genes detected in this family belonged to the SELENOU1 branch. Within fish, we mapped a first well-supported duplication of the SELENOU1 gene in Osteoglossocephalai, yielding SELENOU1a2 ( Figure 7b ). This gene was then lost at the root of the large group of Acanthomorphata, while it converted to Cys homolog in Esox lucius . Prior to both of these events, we traced two additional SELENOU1 duplications in Clupeocephala, yielding the SELENOU1b and SELENOU1c genes. SELENOU1b was then lost independently in three lineages: Otophysi, Esocidae and Atherinomorphae (Supplementary Figure S1). Finally, we detected another two independent well-supported SELENOU1 duplications in Cyprinoidei and Salmonidae, yielding the gene SELENOU1a1b and SELENOU1a1c , respectively ( Figure 7b ). Lastly, within tetrapods, SELENOU1 underwent at least four convergent Sec-to-Cys conversions ( Figure 7c ), specifically in Theria (i.e. placentals and marsupials), Anura (frogs), Testudines (turtles), and Psittaciformes birds (parrots). SELENOW, SELENOV and MIEN1 Selenoprotein W (SELENOW) is a widespread glutathione-dependent antioxidant oxidoreductase implicated in resolution of inflammation, with reported roles in muscle growth and differentiation, as well as in protecting neurons from oxidative stress during neuronal development [ 40 , 41 ]. Selenoprotein V (SELENOV) is a placental-specific putative oxidoreductase expressed in testis and potentially involved in antioxidant defense and ER stress, whose C-terminus shares substantial sequence similarity to SELENOW [ 42 ]. Our present analyses are consistent with SELENOV emerging in placentals by duplication of SELENOW ( Figure 8a ). It was previously reported that the SELENOV gene was lost in Gorilla gorilla [ 16 ]. By analyzing the species tree annotated with SELENOV gene predictions (Supplementary Figure S1), we pinpointed additional potential losses, which we sought to corroborate by searching RNA datasets (Methods). Due to SELENOV detection in RNA data, we ultimately dismissed all apparent losses except two: the known one in Gorilla gorilla , and a novel one in Odontoceti (toothed whales, including dolphins) ( Figure 1 ). Download figure Open in new tab Figure 8. Phylogeny of the SELENOW and SELENOP families. a, b) Gene tree visualizations of the SELENOW and SELENOOP families, respectively. For tree annotation and legend details, see Figure 2c . Expanded gene trees are available as Supplementary Figures S18 (SELENOW) and S22 (SELENOP). Besides SELENOV , mammals have a single other homolog in this gene family: SELENOW , also referred to as SELENOW1 . Our analyses indicate that this gene was independently lost in Cyclostomata, Chondrichthyes and within Actinopterygii, specifically in Acanthomorphata, resulting in its absence from the majority of extant fish genomes (Supplementary Figure S1). Earlier work pinpointed multiple SELENOW homologs in non-mammalian vertebrates, including two subfamilies named SELENOW2 and MIEN1 (formerly Rdx12) [ 43 ]. Indeed, our analysis recovered these two clades: they were defined by a duplication node supported by ∼35 species, and by performing BLASTP of each sequence, we noted that the majority of one clade matched sequences annotated as SELENOW2, while the other matched MIEN1 annotations. These clades were confidently placed as sister groups, with the SELENOW and SELENOV clade as their closest relative ( Figure 8a ). Within the SELENOW2/MIEN1 group, all genes belonging to invertebrates were placed in the SELENOW2 branch (Supplementary Figure S18). This suggests that SELENOW2 predates vertebrates, while MIEN1 emerged within this lineage. We detected at least one MIEN1 gene in each vertebrate genome, except Cyclostomata wherein it was consistently absent. Therefore, we propose that MIEN1 originated by a SELENOW2 duplication at the root of Gnathostomata. In Amniota, MIEN1 genes carry Cys in place of Sec, implying a Sec-to-Cys conversion in this clade. On the other hand, several Actinopterygian lineages contained multiple Sec-containing MIEN1 copies, which we ascribed to secondary lineage-specific gene duplications, detailed hereafter. Because of poor consistency between the protein tree topology and species taxonomy, here we heavily relied on the species tree to infer gene histories. Cyprinoidei contained three copies; in the protein tree, MIEN1 and MIEN1d exhibited the expected topology for duplication, while MIEN1b was placed closer to the root, possibly due to long branch attraction ( Figure 8 ). We consider most likely that two MIEN1 duplications at the root of Cyprinoidei generated MIEN1b and MIEN1d . Salmonidae also exhibited three copies: MIEN1 , MIEN1e , MIEN1c . We argue that these emerged independently of those in Cyprinoidei, since virtually all other Clupeocephala contained a single copy (Supplementary Figure S1). MIEN1e was found in Salmonidae and also in their outgroup E. lucius , which together form the superorder of Protacanthopterygii. We ascribe the MIEN1e origin to a duplication at the root of this lineage; the protein tree topology is consistent with this scenario, albeit with very low UFB support in the corresponding node (6%). The other copy ( MIEN1c ), in contrast, was detected in Salmonidae only. Surprisingly, this clustered with Cyprinoidei MIEN1 and MIEN1d . We consider that the most likely scenario is that MIEN1c emerged in Salmonidae after the split from E. lucius , then its high divergence confounded its phylogenetic position. Lastly, Poeciliidae contained two copies, MIEN1 and MIEN1f . We ascribed this to a gene duplication at the root of this lineage; the protein tree topology is consistent with this, despite an unexpected clustering with MIEN1 of Chondrichthyes. The SELENOW2 branch of the gene tree includes selenoprotein genes from invertebrates, Cyclostomata, Amphibia, and diverse Actinopterygii. The absence of Amniotes implies a gene loss in this lineage, as previously reported [ 16 ]. In Actinopterygii, on the other hand, the history of this gene is more obscure and difficult to solve. SELENOW2 occurs in a scattered pattern across the species tree, apparently implying many independent losses, which may seem unlikely (Supplementary Figure S1). Therefore, we first set to assess the quality of our tree-based orthology assessment using a complementary syntenic analysis. We observed limited but detectable syntenic conservation around SELENOW1 across Actinopterygii and Tetrapoda, shown by consistent nearby presence of EHD2/EHD2A (Supplementary Figure S19). Similarly, genes in the SELENOW2 tree branch are consistently located in the vicinity of the TRPM7 gene in all tested Actinopterygii and Amphibia (Supplementary Figure S20). As per MIEN1, syntenic conservation was clear within Actinopterygii ( SLC4A1A, UBTF, ATXN7I3/ATXN7I3A ) and within Amniota ( IKZF3 , PGAP3, GRB7, ERBB2 ), but not in-between (Supplementary Figure S21). No common nearby genes were observed between SELENOW1 , SELENOW2 and MIEN1 . Altogether, these analyses corroborated the true orthology of these protein clades. Therefore, we proceeded to map SELENOW2 losses within Actinopterygii by examining the species tree annotated with gene presence (Supplementary Figure S1) and seeking validation of putative losses in RNA data. Strikingly, we identified 13 independent losses (after dismissing two due to SELENOW2 detection in RNA sequences; Supplementary Table T2), displayed in Figure 1 . Although additional SELENOW2 -like RNA sequences were recovered in certain genomes (e.g., Oryzias melastigma, Ictalurus punctatus ), further inspection indicated that these sequences likely represented contamination, and they were therefore excluded. SELENOP Selenoprotein P (SELENOP) is the only mammalian selenoprotein with multiple Sec residues, e.g. 10 in human and M. musculus , due to its Sec-rich C-terminal tail. SELENOP is a key player in selenium transport and distribution, serving as the primary selenium carrier in the bloodstream, delivering selenium to peripheral tissues via receptor-mediated uptake. We recently characterized the evolution of the SELENOP gene family, showing that its Sec content can vary by two orders of magnitude across Metazoa [ 44 ]. Its C-terminal tail is challenging to identify using genomic data due to its fast evolution, so that here we mostly focused on gene-level detection. As expected, we identified two phylogenetic clades of vertebrate genes, multi-Sec SELENOP/SELENOP1 , and single-Sec SELENOP2/SELENOPB . While the former is reportedly ubiquitous in vertebrate genomes, the latter was lost in Eutheria [ 44 ]. Moreover, SELENOP2 was converted to a Cys homolog in Anura. Our present analysis ( Figure 8b , Supplementary Figure S22) confirmed these findings, and offered additional insights into early vertebrate evolution. Invertebrate SELENOP homologs are consistently recovered outside of the clade containing both vertebrate paralogs, supporting the hypothesis that the SELENOP1/SELENOP2 duplication arose at the root of vertebrates. Most fish genomes contained the two Sec- SELENOP genes; a notable exception was the Selachii division within Chondrichthyes (i.e. sharks), where SELENOP2 was converted to Cys homolog. In Salmonidae, we traced one lineage-specific gene duplication for each paralog, yielding SELENOP1b and SELENOP2b . Like most fish, Cyclostomata contains two genes, one within the SELENOP1 branch and the other within SELENOP2 . Interestingly, the latter displays a unique CXXU motif (rather than the more typical UXXC) in its N-terminal thioredoxin-like domain. Moreover, though SELENOP2 of other vertebrates encodes a single Sec, in Cyclostomata it exhibits an extraordinary expansion of Sec content: it harbors 162 and 139 in-frame UGA codons in the lampreys Lethenteron reissneri and Lampetra fluviatilis , respectively. These genes were detected in genome assemblies, and their sequence was confirmed by highly similar matches in transcriptome shotgun assembly (TSA) data of other Cyclostomata, spanning all the putative transcript (Supplementary Document D1). Analogously to other SELENOP genes [ 44 ], the vast majority of putative Sec residues are encoded within an expanded modular C-terminal region, and two SECIS elements are detected downstream. We argue that all these in-frame UGAs are translated as Sec: (i) homologous SECIS elements from human, M. musculus and D. rerio SELENOP have been shown to support processive insertion at multiple Sec-UGA sites in a variety of in vitro and in vivo models [ 4 , 44 ]; and (ii) the lamprey coding sequence contains more than 130 stop codons exclusively of the UGA type (rather than a mixture with UAA or UAG), providing strong evidence of selective pressure to preserve the same peculiar gene architecture that is observed in hundreds of multi-Sec SELENOP genes across Metazoa [ 44 ]. With this record, lampreys surpasses all other metazoans in the number of Sec-UGAs in SELENOP , including the previous high of 132 reported in bivalves [ 44 ]. Between the SECISes, we noted the presence of a distinctive region rich in CA dinucleotides (∼1500-1800nt) of unknown significance (Supplementary Document D1). Other Salmonidae-specific duplications: SELENOE, SELENOF, SELENOH Our analysis revealed a substantial selenoproteome expansion specific to the lineage of Salmonidae, which may suggest an increased functional reliance of this clade on selenium. Besides the aforementioned expansions in the GPX, DIO, TXNRD, MSRB, SELENOT, SELENOU and SELENOW protein families ( Figure 1 ), we detected well-supported gene duplications of SELENOE and SELENOF , yielding SELENOEb and SELENOFb , respectively, and another two for SELENOH , yielding SELENOHb and SELENOHc ( Figure 9 , Supplementary Figures S23, S24, S25). Briefly, SELENOF is a ER-resident selenoprotein involved in redox quality control [ 45 ], recently pinpointed as tumor suppressor [ 46 , 47 ]. Our results show that SELENOF is ubiquitous to vertebrates. On the other hand, SELENOE (formerly Fep15), a functionally uncharacterized selenoprotein distantly related to SELENOF [ 12 ], was found as selenoprotein only in fishes: Anura carry a Cys-homolog, while the rest of tetrapods (Amniota) lost the gene altogether. Notably, we mapped the origin of the SELENOE gene to the root of Gnathostomata, since we did not find it in Cyclostomata nor in the non-vertebrate outgroups. Finally, SELENOH is a widespread nucleolar oxidoreductase that regulates redox homeostasis and suppresses DNA damage through the p53 pathway, and it is implicated in cancer proliferation [ 48 – 50 ]. Apart from the Salmonidae-specific duplications, no other gene evolution event was detected for SELENOE , SELENOF , and SELENOH , with the only exception of the loss of SELENOH in Cyclostomata, and in independently in Callorhinchus milii . Download figure Open in new tab Figure 9. Phylogeny of SELENOE, SELENOF, and SELENOH. a, b, c) Gene tree visualizations of the SELENOE, SELENOF, and SELENOH families, respectively. For tree annotation and legend details, see Figure 2c . Expanded gene trees are available as Supplementary Figure S23 (SELENOE), S24 (SELENOF), and S25 (SELENOH). It is worth mentioning that, because all the Salmonidae genomes in our dataset belong to its major subgroup, Salmoninae, we had initially parsimoniously mapped all the gene duplications described above to Salmoninae. However, inspection of intron positions relative to protein sequence alignments ( Figures 2 , 3 , 4 , 6 ) revealed that they were conserved across these lineage-specific duplications. This pattern is consistent with DNA-based gene duplication rather than RNA-based retrotransposition. Consequently, the whole-genome duplication (WGD) previously reported at the root of Salmonidae [ 51 ] provides the most plausible explanation for the origin of these genes, leading us to map these events to Salmonidae as a whole rather than exclusively to Salmoninae. Other families: SEPHS2, SELENOL, SELENOK, SELENOM, SELENON, SELENOS Hereafter, we briefly summarize the remaining vertebrate selenoproteome families, wherein our present analysis recapitulated earlier results without novel findings. As already introduced, SEPHS2 (aka SPS2) is a selenoprotein whose function is critical for Sec biosynthesis. As expected [ 16 , 22 ], we identified the SEPHS2 gene in all vertebrate lineages. (As discussed in the first section of the Results, the gene appears to be absent from many assemblies of birds, but we attribute this to a sequencing bias rather than true gene loss). The SEPHS2 gene is intronless only in genomes of Eutheria (placentals). As previously shown [ 16 ], the ancestral intron-containing SEPHS2 gene (referred to as SEPHS2a ) duplicated by retrotransposition at the root of Mammalia, yielding an intron-less copy referred to as SEPHS2b . Both copies are present in extant genomes of Marsupialia, while SEPHS2a was then lost in Eutheria. Our results confirmed these findings (Supplementary Figures S1, S26). Thus, the sole placental SEPHS2 gene is actually SEPHS2b . Of note, SEPHS2 duplicated at the root of vertebrates, and independently also in various other metazoan lineages, to yield a non-Sec/non-Cys paralog known as SEPHS1/SPS1 [ 22 ]. The vertebrate SEPHS1 gene carries threonine in place of Sec. This kind of conversion is highly unusual for selenoproteins, since Sec is typically catalytic and it is highly dissimilar from canonical amino acids other than Cys. Indeed, SEPHS1 appears to lack the enzymatic activity of SEPHS2 (selenophosphate synthesis from selenide), and instead has a function in gene expression-mediated control of redox homeostasis [ 6 , 52 ]. SELENOL is a peculiar selenoprotein featuring two nearby Sec residues that can form a diselenide bond. It features a peroxiredoxin-like motive suggestive of a redox-related function [ 13 ]. As expected [ 16 ], we detected the SELENOL gene in nearly all vertebrates except Tetrapoda, confirming a gene loss at the root of this lineage (Supplementary Figure S1). The remaining selenoprotein families were detected in a single Sec-containing copy in all investigated genomes (with the only exception of SELENOM , lost in Hyperoartia; Supplementary Figure S1), barring pseudogenes and spurious absence attributed to imperfect assembly quality. Curiously, all these single-copy selenoproteins localize to the ER. They include: SELENOK, involved in calcium homeostasis and resolution of ER stress [ 53 ]; SELENOM and SELENOS, involved in unfolded protein responses [ 54 ]; and SELENON, involved in calcium and redox homeostasis [ 55 ]. Shifts in selenoproteome-to-proteome ratio across vertebrate evolution Our analyses revealed that the vertebrate selenoproteome expanded through many gene duplications in fish lineages. The conservation of intron structures and the phylogenetic correspondence with documented WGD suggests that many of these gene duplications resulted from such events ( Figure 1 ). This raises the question of whether such selenoproteome expansions are adaptive, or are neutral consequences of having a larger proteome. To gain insights in this matter, we compared the size of our predicted selenoproteome with that of the annotated proteome for each species ( Figure 10 ). We fit a zero-intercept linear model separately for each phylogenetic clade of interest (Mammalia, Sauropsida, Chondrichthyes, Salmonidae, Cyprinoidei, and non-Salmonidae non-Cyprinoidei Actinopterigyii), with the hypothesis that the slope (i.e. the selenoproteome-to-proteome ratio) may differ among clades. We found that Actinopterygii have significantly larger selenoproteome-to-proteome ratios than Mammalia ( Figure 10 ). Importantly, aquatic environments have long been hypothesized to favor selenoprotein usage [ 56 ], and our present analysis mostly supports this pattern. Yet, some lineages diverge from this trend: aquatic Salmonidae and Cyprinoidei, which have the biggest selenoproteomes across vertebrates, show low ratios (albeit with substantial statistical uncertainty, due to the few available genomes and their spread in apparent proteome size; see their wide confidence intervals in Figure 10 ). On the other hand, Sauropsida exhibits relatively high ratios despite being land-dwelling. Overall, this data is best explained by an ancestral state of rich selenoproteome-to-proteome ratio, subsequently strongly decreased in Mammalia, and possibly slightly increased in Actinopterygii. Download figure Open in new tab Figure 10. Genomic correlation between annotated proteome size and the number of predicted selenoproteins. Each point represents a species and is color-labelled according to lineage. Trend lines corresponding to zero-intercept linear models fit separately for each lineage (Methods) are displayed. Shading indicates the 95% confidence interval on the slope. The colored text provides the color key and includes the average selenoproteome-to-proteome ratio as percentage. To formally assess the impact of habitat on selenoproteome size, we fitted a series of regression models relating proteome size (denoted as P) to selenoproteome size (S), with habitat (H: aquatic vs. terrestrial) included either as an additive term or as an interaction with proteome size. Because species are not statistically independent, we implemented phylogenetic generalized least squares under both Brownian motion (neutrality) and Ornstein–Uhlenbeck (balancing selection) models, thus incorporating the species tree as a covariance structure. We also tested alternative codings of habitat in which, besides fish, Cetacea and Amphibia were classified as aquatic. Additionally, we considered models in which only Cetacea were classified as aquatic, while Amphibia were treated as terrestrial. Across all tested models, the effect of aquatic habitat was highly significant ( Table 1 ), consistently indicating an enrichment of the selenoproteome in aquatic lineages. Models with the interaction term (S ∼ P + H*P + 0) provided a better fit than additive models (S ∼ P + H + 0), consistent with aquatic species having a proportionally larger selenoproteome relative to proteome size. Likelihood ratio tests further supported the habitat effect, as each model outperformed its corresponding null model lacking H ( Table 1 ). View this table: View inline View popup Table 1. Analysis of selenoproteome-to-proteome ratio in aquatic vs terrestrial vertebrates. The table contains results of different regression models fitted with data from 263 vertebrates (Methods). Two formulas were evaluated: an additive model (S ∼ P + H + 0) and an interaction model (S ∼ P + H*P + 0*), where S is the number of predicted selenoproteins (Selenoprofiles), P the number of annotated proteins (Ensembl), and H a binary variable for aquatic vs. terrestrial habitat. The “+0” denotes that the intercept was constrained to zero. Models included simple linear regression (LM) and phylogenetic generalized least squares (PGLS) under either Brownian motion (BM, neutrality) or Ornstein–Uhlenbeck (OU, balancing selection). All fish were considered aquatic; in some models, Cetacea and/or Amphibia were also classified as aquatic (see the third column). The fourth column reports the P-value of the habitat coefficient, and the fifth column the likelihood ratio test P-value comparing each model to its corresponding null model without H (S ∼ P + 0). DISCUSSION The recent acceleration of genome sequencing and the rise of global biodiversity initiatives has generated an unprecedented abundance of genomic data across vertebrate lineages. This surge presents a unique opportunity to revisit fundamental questions about gene evolution, functional diversity, and lineage-specific adaptations. Here, we leveraged this genomic wealth to provide an updated view of the composition and evolution of the vertebrate selenoproteome, obtained through analyses of hundreds of genomes and meticulous phylogenetic reconstruction which greatly expanded upon previous endeavors. We reinforced and refined existing findings, and uncovered new genes and evolutionary events. Our reconstruction of the ancestral vertebrate selenoproteome consists of 27 selenoprotein genes representing all the 19 known vertebrate Sec-encoding gene families ( Figure 1 ). Our conclusions mostly overlap with our earlier analysis [ 16 ], with the exceptions of GPX2 and SELENOE (both now mapped to Gnathostomata rather than Vertebrata), and GPX3b (Vertebrata rather than Actinopterygii). Besides, we now consider TXNRD3 as the ancestral gene in the TXNRD1/TXNRD3 group, with TXNRD1 emerging in Gnathostomata. We also concluded that these selenoproteins emerged specifically at the root of Vertebrata: GPX3b, DIO2, DIO3, SELENOI, SELENOP2 . Among these, SELENOI stands out in that the whole family appeared at the root of vertebrates: there is no homolog that includes the Sec position, regardless of the aligned amino acid, outside of vertebrates. On the other hand, our present work substantially expands the catalog of selenoproteins in fish. Aquatic environments have long been hypothesized to favor selenoprotein usage, a trend first suggested by Lobanov et al. from limited genomic data [ 56 ]. Subsequent analyses further corroborated this model by showing relaxed purifying selection on selenoprotein genes in terrestrial vertebrates [ 57 ], alongside reduced Sec content in SELENOP, an established proxy for selenium demand [ 44 , 57 ]. Our present analysis of the selenoprotein-to-proteome ratio, leveraging the largest dataset to date and phylogeny-aware models, provides further support for this evolutionary pattern. However, we note that the difference between the selenoproteome of aquatic and land-dwelling species appears primarily driven by mammalian loss and actinopterygian expansion. The underlying cause of the selective advantage of Sec in aquatic environments is arguably yet unclear. It has been proposed that land-dwelling organisms must face (i) reduced selenium bioavailability and (ii) increased oxygen exposure [ 56 , 57 ]. While the latter seems intuitively relevant, it contradicts one of the prevalent proposed biochemical rationale for usage of selenium (Sec) over sulphur (Cys) [ 58 , 59 ]: its resistance to (over)oxidation - a trait that would suggest selenoproteins should be more beneficial in terrestrial environments. Moreover, our data shows that selenoproteins belonging to antioxidant families are more prevalent in fish compared to tetrapods. While further research is required to fully resolve this question, we argue that the current evidence better supports selenium bioavailability as the most likely evolutionary driver of Sec usage in aquatic vs land environments. These observations may depict our own selenoproteins as an unstable remnant of an aquatic past. Yet, it is worth noting that (i) land-living tetrapods exhibit substantial purifying selection, preventing Sec to Cys conversions in most genes [ 60 ]; and (ii) the aquatic selenoproteome is far from static, having expanded under putative adaptive forces at the root of Actinopterygii and in multiple of its clades. We found that these duplications occurred primarily in protein families linked to antioxidant defense, either with well-established roles (GPX, MSRB) or with putative/indirect functions (SELENOT, SELENOU, SELENOW). In contrast, no variation was observed in ER-resident proteins with functions in calcium homeostasis (SELENOK, SELENON) or protein quality control / unfolded protein response (SELENOM, SELENOS). While this trend was not absolute - e.g. ER-resident SELENOE and SELENOF duplicated in Salmonidae, and the frequently duplicated SELENOT is also ER-resident - it suggests that selenoproteome expansions in Actinopterygii may have been driven primarily by selection for enhanced antioxidant capacity. While we did not investigate the exact mechanism of gene duplications, we note that, strikingly, 40 selenoprotein duplications out of the 56 detected across vertebrates were mapped to taxonomic nodes corresponding with reported WGD events ( Figure 1 ). This observation reinforces the importance of WGDs in shaping vertebrate proteomes. Among tetrapods, our most striking finding is the recurrence of convergent evolutionary events, including Sec-to-Cys conversions in SELENOU1 and GPX6, as well as independent losses of SELENOV, all of which are supported by robust phylogenetic evidence. It is important to note that, because our reconstruction is based on maximum parsimony, our estimate of such events represents a conservative lower bound. Interestingly, both SELENOV and GPX6 are the “youngest” mammalian selenoproteins and they are specifically expressed in the testis, a tissue recognized for its role in the evolution of new genes [ 61 ]. Our study has some limitations. First, we restricted our search to genes homologous to known selenoprotein families, because existing methods for identifying novel selenoproteins suffer from low specificity [ 62 ]. As a result, we may have overlooked lineage-specific selenoproteins, particularly in fish lineages where selenoproteome expansions are evident. Second, we could only detect events within lineages represented in our dataset. Given the dynamic nature of the Actinopterygii selenoproteome, additional genome sequences from this group will likely reveal further duplications. Third, our analysis assumes the species tree as ground truth, which may not be entirely accurate. For instance, the convergent loss of the same genes in Tetraodontidae, Perciformes, Anabantaria, and Cynoglossus semilaevis ( Figure 1 ) could be explained as a single ancestral loss if the non-Perciform Eupercaria and Carangaria were instead the outgroups of these clades - yet, we found no supporting evidence for this alternative phylogeny in the literature [ 63 ]. Fourth, further data or alternative methodological approaches could refine the evolutionary history of selenoprotein families with the most complex gene dynamics, such as SELENOW and SELENOU, and provide deeper insight into lineages with highly complex genome evolution, such as Cyprininae. Despite these limitations, we anticipate that our characterization of vertebrate selenoproteins will largely hold up over time and provide a meaningful framework for future evolutionary and functional research. MATERIAL AND METHODS Genome sequences and species tree We downloaded all genome assemblies and annotations available in Ensembl version v.108 [ 64 ] comprising 315 genomes, each corresponding to a different species or strain. These included 310 vertebrates, two tunicates, one nematode ( C. elegans ), one insect ( D. melanogaster ), and yeast Saccharomyces cerevisae (here excluded from all analyses). The phylogenetic tree of species was downloaded from Ensembl Compara v.e110 [ 20 ]. To refine our reconstruction of the ancestral vertebrate selenoproteome, we also analyzed NCBI assemblies from species with informative taxonomy; identifiers are provided in Supplementary Table T3. Specifically, we included various early-branching vertebrates: nine cartilaginous fish (Chondrichthyes) and nine jawless vertebrates (Hyperoartia class, belonging to the Cyclostomata superclass); then four amphioxiformes (Branchiostoma), serving as additional close relatives to vertebrates, besides the two tunicates already in Ensembl; and two remote outgroups: Strongylocentrotus purpuratus (Echinodermata, deuterostome) and Strigamia maritima (Arthropoda, protostome). These 24 genomes, together with the 314 Ensembl assemblies, constituted our main dataset. Later, we searched two additional Cyclostomata genomes, belonging to the class of Myxini, whose predictions were included in the protein tree reconstruction of TXNRD. For the rest of families, we distinguished between Cyclostomata, Myxini, and Hyperoartia specific events by inspecting the species tree annotated with selenoprotein predictions (Supplementary Figure S1). For follow up analyses to confirm or dismiss gene evolution events, we then inspected gene predictions on an extended set of 1,172 vertebrate genome assemblies downloaded from NCBI, using the NCBI taxonomy tree as the rough backbone for evolutionary interpretation. Selenoprotein gene prediction Selenoprotein gene prediction was performed using Selenoprofiles [ 19 , 65 , 66 ] version v4.5.0, available at https://github.com/marco-mariotti/selenoprofiles . Selenoprofiles is a computational pipeline to identify members of the known selenoprotein families and related proteins. It uses several homology-based gene prediction programs, and employs a set of manually curated multiple sequence alignments of selenoprotein families to scan genomes for homologues. The profiles we used are available at https://github.com/marco-mariotti/selenoprotein_profiles (v1.3.0). We scanned genome assemblies for all known selenoprotein families in Metazoa. For most analyses, we focused on the predictions labelled as “selenocysteine”, i.e., the genes encoding for selenoproteins, with no apparent pseudogene features (e.g. frameshifts). Phylogenetic reconstruction of selenoprotein families The protein sequences of all Selenoprofiles predictions across Ensembl genomes were collected using the Selenoprofiles join utility, and alignments were computed using ClustalO v.1.2.4 [ 67 ]. These were used as input to phylogenetic reconstruction using IQ-TREE v3.0.1 [ 68 ]. For each selenoprotein family, the best-fitting substitution model was automatically selected using ModelFinderPlus. The selected models are listed in Supplementary Table T4. Phylogenetic trees were inferred under the corresponding substitution model, and the branch support values were estimated using 1,000 ultrafast bootstrap replicates [ 68 ]. We visualized gene trees with the show_tree program in the biotree_tools package v0.0.6 ( https://github.com/marco-mariotti/biotree_tools ), based on the ETE3 python module [ 69 ], specifically adapted for custom visualization in this manuscript. The resulting vectorial images were further annotated manually using the Inkscape software. Sequence and tree manipulations throughout the project were carried out using the programs alignment_tools and tree_tools from the biotree_tools package, all based on ETE3 [ 69 ]. Sequences in the SELENOP gene family were trimmed prior to tree building to remove the highly divergent C-terminal Sec-rich tail, using custom code based on Pyaln v0.1.4 ( https://github.com/marco-mariotti/pyaln ). Analysis of gene evolution events We manually inspected gene and species trees to determine gene evolution events: losses, duplications, and Sec-to-Cys conversions. To detect potential duplications, gene trees were processed using the species overlap algorithm: the number of common species between the left and right branch of every ancestral node was used as a quantitative indicator of a gene duplication in the last common ancestor of those species. Putative duplication nodes were colored in the gene tree visualization, and their size was made proportional to this indicator, facilitating our manual review. A visualization of the species tree together with all gene predictions in each genome was produced using the tree drawer utility available in Selenoprofiles, customized as follows. First, each gene prediction corresponding to a multi-member gene family were assigned to a mammalian-centric homology group, hereafter referred to as gene “subfamily” (e.g. GPX1, GPX2, GPX3 , etc), using the orthology utility of Selenoprofiles. Then, we encoded the expected selenoproteome for every lineage based on literature; e.g. bats, since they are mammals with no further lineage-specific events reported, were expected to carry one Sec-encoding gene for each of GPX1, GPX2, GPX3, GPX4 and GPX6 . In our custom tree drawer visualization (Supplementary Figure S1), we marked each gene prediction based on the deviation from such expectations; e.g., bats were labelled as missing a Sec- GPX6 . This highlighted the presence of extra genes (putative duplications) and missing genes (putative losses). Ultimately, the joint analysis of gene and species trees led us to manually map each event into the species tree, following criteria of maximum parsimony and consistency with the known phylogenetic relationship among species. Lastly, each event was critically evaluated to exclude potential artifacts. We required support by at least two sister species for most events. Single-species events were deemed reliable only if supporting independent data (e.g. RNA sequences) existed. Due to imperfect quality of genome assemblies, we sought confirmation of each individual gene loss using independent sequence data from the lineage of interest, as follows. At this stage, each putative gene loss consisted in the lack of an apparently functional gene (i.e., without mutations introducing frameshifts or in-frame stops) in the set of Selenoprofiles predictions on genome assemblies in our main dataset. First, we inspected the full set of Selenoprofiles predictions for the relevant gene family in an extended set of NCBI genomes from the relevant clade. If the loss was not dismissed, we then selected the protein sequence of the putatively lost gene from a closely related organism, and searched it using TBLASTN at the NCBI web server [ 70 ] in all RNA data available for the clade that included the focal species, i.e. TSAs, Expressed Sequence Tags (ESTs), and Reference RNA sequences (refseq_rna). While absence of transcriptomic data alone is not conclusive evidence of gene loss (e.g. due to tissue-specific expression), we argue that it provides strong support when it is combined with the concomitant absence from multiple genome assemblies in the same clade. Grx-domain analysis in TXNRD genes To investigate the distribution of Grx domains encoded in TXNRD genes, we first collected a set of representative TXNRD protein sequences and scanned them for conserved domains using InterProScan [ 71 ]. We extracted the sequences containing the Grx domain and thus manually curated a high-confidence alignment of Grx domains of TXNRDs. These were used as sequence profiles to search for homologous matches in Ensembl vertebrate genomes using Selenoprofiles. Next, we cross-analyzed this data with those TXNRD gene predictions in the same genomes classified as either TXNRD1 or TXNRD3 by the Selenoprofiles lineage utility. We used a custom script based on pyranges v1 [ 72 ] to assign Grx domain predictions to TXNRD1 and TXNRD3 genes, if located within their extended genomic coordinates (±5,000 nucleotides). Unassigned Grx domains were discarded. Finally, we visualized domain architecture and evolutionary patterns by visualizing data through custom scripts based on the ETE3 library [ 63 ]. Synteny analysis Through custom scripts based on the library pyranges v1 [ 72 ], we selected all gene annotations that overlapped the genomic loci of the gene of interest ( TXNRD1 , TXNRD3 , and SELENOW2/MIEN1 ; Supplementary Figures S7, S8, S19, S20) predicted by Selenoprofiles, extended on each side by 300,000 bp, in a set of representative species. For these syntenic analyses, we used Ensembl v.115 annotations, upon realizing their annotated gene names were improved compared to v.108, which we had used for the rest of our work. We then used python code based on the graphics library pyrangeyes v.1.0.7 ( https://github.com/pyranges/pyrangeyes ) to produce custom visualizations wherein genes with the same name were colored accordingly, allowing to assess their syntenic conservation. To inspect the homologous region in species lacking the gene of interest, we performed an analogous process on a gene that consistently occurred in the its neighborhood in other species ( CHST11 for TXNRD1 ; TRPM for SELENOW2 ; RSPH4A for SELENOW1 ). Synteny representations were manually collated next to the species tree using the Inkscape software. Gene nomenclature We strived to assign gene names as consistent as possible with existing resources and literature. To do so, we consulted the relevant publications on vertebrate selenoprotein evolution and nomenclature [ 73 ], as well as current annotations at Uniprot, Ensembl, and NCBI. The reference recommendation for gene names is to derive them from the corresponding mammalian orthologs [ 73 ]. The genes emerged from non-mammalian duplications —many of them previously unreported— were named by appending alphabetical or numerical suffixes to the source gene, e.g. GPX4b , SELENOW2 . To enhance readability, suffixes are written in lowercase throughout this work. To avoid ambiguity, when creating new gene names we systematically skipped the suffixes “a” and “1” (in favor of “b” or “2”), because these are inconsistently used in the literature to refer to non-mammalian orthologs of single-copy mammalian genes; e.g. SELENOW is sometimes referred to as SELENOW1 in species that also possess the paralog SELENOW2 . Regression models of selenoproteome and proteome sizes Regression models were implemented in R, using libraries stats for linear models and likelihood ratio tests, and phylolm for phylogenetic generalized least squares. A first analysis with zero-intercept linear models ( Figure 10 ) was run with 314 annotated species (Ensembl v.108), used to estimate proteome size (number of unique protein-coding genes), whereas Selenoprofiles was used to estimate selenoproteome size. A second analysis ( Table 1 ) was run with the species subset that had a fully resolved taxonomic tree available in Ensembl Compara (263 genomes), using either linear models or phylogenetic generalized least squares; refer to the text and Table 1 for further details. DECLARATIONS Ethics approval and consent to participate Not applicable Consent for publication Not applicable Availability of data and materials The latest Selenoprofiles code is available at https://github.com/marco-mariotti/selenoprofiles4 and documented at https://selenoprofiles4.readthedocs.io/ . Selenoprotein sequences used in this study split by family can be found in Supplementary Data D2. Competing interests The authors declare that they have no competing interests. Funding MM is funded by grants RYC2019-027746-I, PID2020-115122GA-I00, PID2023-147164NB-I00, and JL-F by CNS2022-135805 and PID2022-137753NA-I00, all these funded by MICIU/AEI /10.13039/501100011033 and by “ESF Investing in your future”, FEDER, UE. MM and JL-F received support from Comissió Interdepartamental de Recerca i Innovació Tecnològica (2021SGR00279). Authors’ contributions MT analysed all the data, performed the evolutionary analyses, prepared all the figures and wrote the draft of the manuscript. JL-F contributed to data interpretation and revised the manuscript. MM supervised the project, secured funding, provided expert interpretation of the results and was a major contributor in writing the manuscript. SUPPLEMENTARY DATA Figure S1. Selenoproteome evolution across vertebrate species. Each row represents a species and each column corresponds to a selenoprotein family. Rectangles represent individual Selenoprofiles predictions, color-coded according according to their Selenoprofiles labels, i.e. the codon found aligned to the Sec position: blue for Sec-UGA,orange for Cys, pink for unaligned). Selected lineages denoted with a colored background are labelled on the left. Figure S2. Presence of genes within the SEPHS2 syntenic block across Sauropsidan genome assemblies. The species tree of Sauropsida (plus platypus, used as outgroup) is displayed on the left. Three columns display the presence of genes SEPSH2 (blue, third column) and the genes consistently located as its neighbors in tetrapods, G6PD (first column) and ikbkg (second column). The third column also shows the presence of SEPHS1 (purple), located in a different chromosomic location, here serving as positive control. Note the scattered lack of G6PD, ikbkg, and SEPHS2 from many bird genomes, attributed to imperfect quality of genome assemblies, as discussed in the text. Figure S3. Gene tree of the Glutathione peroxidase (GPX) family. Subfamilies are color-labeled as follows: GPX4 in orange, GPX4b in yellow, GPX3b in purple, GPX3 in green, GPX6 in pink, GPX5 in light purple, GPX2 in beige, GPX1 in brown, and GPX1b in blue. Refer to Figures 2 and 3 for details. Figure S4. Gene tree of the Iodothyronine deiodinase (DIO) family. Subfamilies are color-labeled as follows: DIO1 in orange, DIO2 in yellow, DIO3 in green and DIO3b in purple. See also Figures 2c (layout explanation) and 4 (compressed tree). Figure S5. Iodothyronine deiodinase (DIO) evolution across Salmonidae species. Each row represents a species and each column corresponds to a selenoprotein family. Nodes represent individual Selenoprofiles predictions and are color-coded according to the codon aligned at the Sec position: blue for Sec-UGA, orange for Cys. Figure S6. Gene tree of the Thioredoxin reductase (TXNRD) family. Subfamilies are color-labeled as follows: TXNRD1 in green, TXNRD2 in orange, TXNRD3 in yellow. See also Figures 2c (layout explanation) and 5 (compressed tree). Figure S7. Analysis of synteny of TXNRD1 across vertebrates. The figure displays the genomic neighborhoods of TXNRD1 and/or CHST11 (see Methods) in representative genomes. Same-colored gene structures share the same annotated gene name. The species tree is shown on the left. Species highlighted in grey do not contain the TXNRD1 gene. Figure S8. Analysis of synteny of TXNRD3 across vertebrates. The figure displays the genomic neighborhoods of TXNRD3 and/or CHST13 (see Methods) in representative genomes. Same-colored gene structures share the same annotated gene name. The species tree is shown on the left. Species highlighted in grey do not contain the TXNRD1 gene. Figure S9. Vertebrate species tree representing the presence/absence of Grx domain in TXNRD1/TXNRD3 sequences. Colors are assigned according to the TXNRD gene tree. Figure S10. Gene tree of the Selenoprotein J (SELENOJ) family. Subfamilies are color-labeled as follows: SELENOJ in green and SELENOJ2 in yellow. See also Figures 2c (layout explanation) and 6 (compressed tree). Figure S11. Selenoprotein J (SELENOJ) evolution across Actinopterygii species. Each row represents a species and each column corresponds to a selenoprotein family. Nodes represent individual Selenoprofiles predictions and are color-coded according to their amino acid content at the UGA position: blue for Sec-containing sequences, orange for Cys-containing sequences, and pink for other classifications inferred by Selenoprofiles. Figure S12. Gene tree of the Selenoprotein O (SELENOO) family. Subfamilies are color-labeled as follows: SELENOO in green and SELENOO2 in orange. Nodes highlighted in brown and light purple indicate sequences from Tetraodontidae and Oryziinae species, respectively. See also Figures 2c (layout explanation) and 6 (compressed tree). Figure S13. Selenoprotein O (SELENOO) evolution across Actinopterygii species. Each row represents a species and each column corresponds to a selenoprotein family. Nodes represent individual Selenoprofiles predictions and are color-coded according to their amino acid content at the UGA position: blue for Sec-containing sequences, orange for Cys-containing sequences, and pink for other classifications inferred by Selenoprofiles. Figure S14. Gene tree of the Methionine sulfoxide reductase B (MSRB) family. See also Figures 2c (layout explanation) and 6 (compressed tree). Figure S15. Methionine sulfoxide reductase B (MSRB) evolution across Actinopterygii species. Each row represents a species and each column corresponds to a selenoprotein family. Nodes represent individual Selenoprofiles predictions and are color-coded according to their amino acid content at the UGA position: blue for Sec-containing sequences, orange for Cys-containing sequences, and pink for other classifications inferred by Selenoprofiles. Figure S16. Gene tree of the Selenoprotein T (SELENOT) family. Subfamilies are color-labeled as follows: SELENOT1 in green and SELENOT2 in yellow. Nodes highlighted in light blue indicate sequences from Salmonidae species, while lightpink and grey indicate sequences from Cyprinoidei and Otophysi species, respectively. See also Figures 2c (layout explanation) and 7 (compressed tree). Figure S17. Gene tree of the Selenoprotein U (SELENOU) family. Subfamilies are color-labeled as follows: SELENOU1a in yellow, SELENOU1a2 in orange, SELENOU1b in green, SELENO1c in purple and Cysteine-homolog PRLX2A in beige. Nodes highlighted in light blue indicate sequences from Salmonidae species, while lightpink indicate sequences from Cyprinoidei. See also Figures 2c (layout explanation) and 7 (compressed tree). Figure S18. Gene tree of the Selenoprotein V/W (SELENOV/W) family. Subfamilies are color-labeled as follows: SELENOW1 from Actinopterygii species in green, SELENOW1 from tetrapoda species in beige, SELENOV in purple, SELENOW2a in yellow and SELENOW2c in orange. Nodes highlighted in light blue indicate sequences from Salmonidae species, while light pink and light grey indicate sequences from Cyprinoidei and Cyprinodontoidei species, respectively. See also Figures 2c (layout explanation) and 8 (compressed tree). Figure S19. Analysis of synteny of SELENOW1 across vertebrates. The figure displays the genomic neighborhoods of SELENOW1 and/or rsph4a (see Methods) in representative genomes. Same-colored gene structures share the same annotated gene name. The species tree is shown on the left. Species highlighted in grey do not contain SELENOW1 gene. Figure S20. Analysis of synteny of SELENOW2 across vertebrates. The figure displays the genomic neighborhoods of SELENOW2 and/or TRPM7 (see Methods) in representative genomes. Same-colored gene structures share the same annotated gene name. The species tree is shown on the left. Species highlighted in grey do not contain SELENOW2 gene. Figure S21. Analysis of synteny of MIEN1 across vertebrates. The figure displays the genomic neighborhoods of MIEN1 (see Methods) in representative genomes. Same-colored gene structures share the same annotated gene name. The species tree is shown on the left. Figure S22. Gene tree of the Selenoprotein P (SELENOP) family. Nodes highlighted in light blue indicate sequences from Salmonidae species. See also Figures 2c (layout explanation) and 8 (compressed tree). Figure S23. Gene tree of the Selenoprotein E (SELENOE) family. See also Figures 2c (layout explanation) and 9 (compressed tree). Figure S24. Gene tree of the Selenoprotein F (SELENOF) family. Nodes highlighted in light blue indicate sequences from Salmonidae species See also Figures 2c (layout explanation) and 9 (compressed tree). Figure S25. Gene tree of the Selenoprotein H (SELENOH) family. Nodes highlighted in light blue indicate sequences from Salmonidae species. See also Figures 2c (layout explanation) and 9 (compressed tree). Figure S26. Gene tree of the SEPHS family. See also Figure 2c (layout explanation). Table S1. Comparison of Glutathione Peroxidase (GPX) gene nomenclature between Xue et al. and this study. The table lists the gene names assigned to GPX family members in Xue et al. and the corresponding nomenclature used in this study (Ticó et al.). The third column provides relevant notes on differences in naming, such as updated classification, subfamily assignment, or gene duplication interpretations. Table T2. Detection of SELENOW1, SELENOW2 and MIEN1 transcripts across vertebrate species. Boolean values denote presence (True) or absence (False) of RNA-derived sequences assigned to each family based on similarity scoring against the curated reference dataset extracted from the SELENOW gene tree. Corresponding taxonomic lineage is provided for each species. Table T3. List of NCBI species and their corresponding genome assembly accession IDs used in this study. Table T4. Substitution models selected by IQ-TREE for phylogenetic tree reconstruction in each selenoprotein family. Document D1. Sequence analysis of SELENOP2 in lampreys . The document contains the sequences of SELENOP2 genes in two Cyclostomata, obtained via Selenoprofiles in genome data, then corroborated by RNA sequences. SECIS elements were predicted using SECISearch3. Document D2. Protein sequences of all selenoproteins predicted and analyzed in this study. For details on file structure and content, please refer to the accompanying README file. Acknowledgements Not applicable. Funder Information Declared Ministerio de Universidades, MICIU/AEI /10.13039/501100011033 and “ESF Investing in your future”, , RYC2019-027746-I , PID2020-115122GA-I00 , PID2023-147164NB-I00 , CNS2022-135805 , PID2022-137753NA-I00 Comissió Interdepartamental de Recerca i Innovació Tecnològica , 2021SGR00279 Footnotes Rebuilt all trees using IQ-tree, revisited gene histories. REFERENCES 1. ↵ Labunskyy VM , Hatfield DL , Gladyshev VN . Selenoproteins: Molecular Pathways and Physiological Roles . Physiol Rev . 2014 ; 94 : 739 . doi: 10.1152/PHYSREV.00039.2013 . OpenUrl CrossRef PubMed 2. Mariotti M , Gladyshev VN . Selenocysteine-containing proteins . Redox Chemistry and Biology of Thiols . 2022 ;: 405 – 21 . doi: 10.1016/B978-0-323-90219-9.00012-1 . OpenUrl CrossRef 3. ↵ Fairweather-Tait SJ , Bao Y , Broadley MR , Collings R , Ford D , Hesketh JE , et al. Selenium in human health and disease . Antioxid Redox Signal . 2011 ; 14 : 1337 – 83 . doi: 10.1089/ARS.2010.3275 . OpenUrl CrossRef PubMed Web of Science 4. ↵ Howard MT , Copeland PR . New Directions for Understanding the Codon Redefinition Required for Selenocysteine Incorporation . Biol Trace Elem Res . 2019 ; 192 : 18 – 25 . doi: 10.1007/S12011-019-01827-Y . OpenUrl CrossRef PubMed 5. ↵ Copeland PR , Howard MT . Ribosome Fate during Decoding of UGA-Sec Codons . Int J Mol Sci . 2021 ; 22 . doi: 10.3390/IJMS222413204 . OpenUrl CrossRef 6. ↵ Manta B , Makarova NE , Mariotti M . The selenophosphate synthetase family: A review . Free Radic Biol Med . 2022 ; 192 : 63 – 76 . doi: 10.1016/J.FREERADBIOMED.2022.09.007 . OpenUrl CrossRef PubMed 7. ↵ Flohé L . The labour pains of biochemical selenology: The history of selenoprotein biosynthesis . Biochimica et Biophysica Acta (BBA) - General Subjects . 2009 ; 1790 : 1389 – 403 . doi: 10.1016/j.bbagen.2009.03.031 . OpenUrl CrossRef PubMed 8. ↵ Davy T , Castellano S . The genomics of selenium: Its past, present and future . Biochim Biophys Acta Gen Subj . 2018 ; 1862 : 2427 – 32 . doi: 10.1016/j.bbagen.2018.05.020 . OpenUrl CrossRef 9. ↵ Kryukov G V ., Castellano S , Novoselov S V ., Lobanov A V ., Zehtab O , Guigó R , et al. Characterization of mammalian selenoproteomes . Science . 2003 ; 300 : 1439 – 43 . doi: 10.1126/SCIENCE.1083516 . OpenUrl Abstract / FREE Full Text 10. ↵ Castellano S , Novoselov S V ., Kryukov G V ., Lescure A , Blanco E , Krol A , et al. Reconsidering the evolution of eukaryotic selenoproteins: a novel nonmammalian family with scattered phylogenetic distribution . EMBO Rep . 2004 ; 5 : 71 – 7 . doi: 10.1038/SJ.EMBOR.7400036 . OpenUrl Abstract / FREE Full Text 11. ↵ Castellano S , Lobanov A V ., Chapple C , Novoselov S V ., Albrecht M , Hua D , et al. Diversity and functional plasticity of eukaryotic selenoproteins: Identification and characterization of the SelJ family . Proc Natl Acad Sci U S A . 2005 ; 102 : 16188 – 93 . doi: 10.1073/PNAS.0505146102/SUPPL_FILE/05146FIG9.JPG . OpenUrl Abstract / FREE Full Text 12. ↵ Novoselov S V ., Hua D , Lobanov A V ., Gladyshev VN . Identification and characterization of Fep15, a new selenocysteine-containing member of the Sep15 protein family . Biochem J . 2006 ; 394 Pt 3 : 575 – 9 . doi: 10.1042/BJ20051569 . OpenUrl Abstract / FREE Full Text 13. ↵ Shchedrina VA , Novoselov S V ., Malinouski MY , Gladyshev VN . Identification and characterization of a selenoprotein family containing a diselenide bond in a redox motif . Proc Natl Acad Sci U S A . 2007 ; 104 : 13919 – 24 . doi: 10.1073/PNAS.0703448104/SUPPL_FILE/03448TABLE4.PDF . OpenUrl Abstract / FREE Full Text 14. ↵ Jiang L , Ni J , Liu Q . Evolution of selenoproteins in the metazoan . BMC Genomics . 2012 ; 13 . doi: 10.1186/1471-2164-13-446 . OpenUrl CrossRef PubMed 15. ↵ Jiang L , Liu Q , Ni J . In silico identification of the sea squirt selenoproteome . BMC Genomics . 2010 ; 11 : 289 . doi: 10.1186/1471-2164-11-289 . OpenUrl CrossRef PubMed 16. ↵ Mariotti M , Ridge PG , Zhang Y , Lobanov A V ., Pringle TH , Guigo R , et al. Composition and evolution of the vertebrate and mammalian selenoproteomes . PLoS One . 2012 ; 7 . doi: 10.1371/JOURNAL.PONE.0033066 . OpenUrl CrossRef 17. ↵ Lewin HA , Robinson GE , Kress WJ , Baker WJ , Coddington J , Crandall KA , et al. Earth BioGenome Project: Sequencing life for the future of life . Proc Natl Acad Sci U S A . 2018 ; 115 : 4325 – 33 . doi: 10.1073/PNAS.1720115115 . OpenUrl Abstract / FREE Full Text 18. ↵ Mc Cartney AM , Formenti G , Mouton A , De Panis D , Marins LS , Leitão HG , et al. The European Reference Genome Atlas: piloting a decentralised approach to equitable biodiversity genomics. npj Biodiversity 2024 3:1. 2024 ; 3 : 1 – 17 . doi: 10.1038/s44185-024-00054-6 . OpenUrl CrossRef 19. ↵ Ticó M , Sullivan E , Guigó R , Mariotti M . Overcoming the widespread flaws in the annotation of vertebrate selenoprotein genes in public databases . PLoS Comput Biol . 2026 ; 22 : e1013885 . doi: 10.1371/journal.pcbi.1013885 . OpenUrl CrossRef PubMed 20. ↵ Vilella AJ , Severin J , Ureta-Vidal A , Heng L , Durbin R , Birney E . EnsemblCompara GeneTrees: Complete, duplication-aware phylogenetic trees in vertebrates . Genome Res . 2009 ; 19 : 327 – 35 . doi: 10.1101/GR.073585.107 . OpenUrl Abstract / FREE Full Text 21. ↵ Huerta-Cepas J , Szklarczyk D , Heller D , Hernández-Plaza A , Forslund SK , Cook H , et al. eggNOG 5.0: a hierarchical, functionally and phylogenetically annotated orthology resource based on 5090 organisms and 2502 viruses . Nucleic Acids Res . 2019 ; 47 : D309 – 14 . doi: 10.1093/NAR/GKY1085 . OpenUrl CrossRef PubMed 22. ↵ Mariotti M , Santesmasses D , Capella-Gutierrez S , Mateo A , Arnan C , Johnson R , et al. Evolution of selenophosphate synthetases: emergence and relocation of function through independent duplications and recurrent subfunctionalization . Genome Res . 2015 ; 25 : 1256 – 67 . doi: 10.1101/GR.190538.115 . OpenUrl Abstract / FREE Full Text 23. ↵ Zhang Y , Jin J , Huang B , Ying H , He J , Jiang L . Selenium Metabolism and Selenoproteins in Prokaryotes: A Bioinformatics Perspective . Biomolecules 2022 , Vol 12 , Page 917 . 2022;12:917. doi: 10.3390/BIOM12070917 . OpenUrl CrossRef PubMed 24. ↵ Botero-Castro F , Figuet E , Tilak MK , Nabholz B , Galtier N . Avian Genomes Revisited: Hidden Genes Uncovered and the Rates versus Traits Paradox in Birds . Mol Biol Evol . 2017 ; 34 : 3123 – 31 . doi: 10.1093/MOLBEV/MSX236 . OpenUrl CrossRef PubMed 25. ↵ Chapple CE , Guigó R . Relaxation of selective constraints causes independent selenoprotein extinction in insect genomes . PLoS One . 2008 ; 3 . doi: 10.1371/journal.pone.0002968 . OpenUrl CrossRef PubMed 26. ↵ Chipman AD , Ferrier DEK , Brena C , Qu J , Hughes DST , Schröder R , et al. The First Myriapod Genome Sequence Reveals Conservative Arthropod Gene Content and Genome Organisation in the Centipede Strigamia maritima . PLoS Biol . 2014 ; 12 : e1002005 . doi: 10.1371/journal.pbio.1002005 . OpenUrl CrossRef PubMed 27. ↵ Mariotti M . Selenocysteine Extinctions in Insects . 2016 ;: 113 – 40 . doi: 10.1007/978-3-319-24244-6_5 . OpenUrl CrossRef 28. ↵ Toppo S , Vanin S , Bosello V , Tosatto SCE . Evolutionary and structural insights into the multifaceted glutathione peroxidase (Gpx) superfamily . Antioxid Redox Signal . 2008 ; 10 : 1501 – 13 . doi: 10.1089/ARS.2008.2057 . OpenUrl CrossRef PubMed Web of Science 29. ↵ Xue Y , Chen L , Li B , Xiao J , Wang H , Dong C , et al. Genome-wide mining of gpx gene family provides new insights into cadmium stress responses in common carp (Cyprinus carpio) . Gene . 2022 ; 821 : 146291 . doi: 10.1016/J.GENE.2022.146291 . OpenUrl CrossRef 30. ↵ Darras VM , Van Herck SLJ . Iodothyronine deiodinase structure and function: from ascidians to humans . J Endocrinol . 2012 ; 215 : 189 – 206 . doi: 10.1530/JOE-12-0204 . OpenUrl Abstract / FREE Full Text 31. ↵ Lorgen M , Casadei E , Król E , Douglas A , Birnie MJ , Ebbesson LOE , et al. Functional divergence of type 2 deiodinase paralogs in the Atlantic salmon . Curr Biol . 2015 ; 25 : 936 – 41 . doi: 10.1016/J.CUB.2015.01.074 . OpenUrl CrossRef PubMed 32. ↵ Arnér ESJ . Focus on mammalian thioredoxin reductases--important selenoproteins with versatile functions . Biochim Biophys Acta . 2009 ; 1790 : 495 – 526 . doi: 10.1016/J.BBAGEN.2009.01.014 . OpenUrl CrossRef PubMed 33. ↵ Dou Q , Turanov AA , Mariotti M , Hwang JY , Wang H , Lee SG , et al. Selenoprotein TXNRD3 supports male fertility via the redox regulation of spermatogenesis . Journal of Biological Chemistry . 2022 ; 298 . doi: 10.1016/J.JBC.2022.102183/ATTACHMENT/F4C4EF7D-2A2A-49B7-B3BB-B4E7E586A468/MMC3.PDF . OpenUrl CrossRef 34. ↵ Huerta-Cepas J , Bueno A , Dopazo J , Gabaldón T . PhylomeDB: a database for genome-wide collections of gene phylogenies . Nucleic Acids Res . 2008 ; 36 suppl_1 : D491 – 6 . doi: 10.1093/NAR/GKM899 . OpenUrl CrossRef PubMed Web of Science 35. ↵ Fata F , Gencheva R , Cheng Q , Lullo R , Ardini M , Silvestri I , et al. Biochemical and structural characterizations of thioredoxin reductase selenoproteins of the parasitic filarial nematodes Brugia malayi and Onchocerca volvulus . Redox Biol . 2022 ; 51 . doi: 10.1016/j.redox.2022.102278 . OpenUrl CrossRef 36. ↵ Melo LMN , Sabatier M , Ramesh V , Szylo KJ , Fraser CS , Pon A , et al. Selenoprotein O Promotes Melanoma Metastasis and Regulates Mitochondrial Complex II Activity . Cancer Res . 2025 ; 85 : 942 – 55 . doi: 10.1158/0008-5472.CAN-23-2194 . OpenUrl CrossRef 37. ↵ Sreelatha A , Yee SS , Lopez VA , Park BC , Kinch LN , Pilch S , et al. Protein AMPylation by an Evolutionarily Conserved Pseudokinase . Cell . 2018 ; 175 : 809 – 821 .e19. doi: 10.1016/J.CELL.2018.08.046 . OpenUrl CrossRef PubMed 38. ↵ Tarrago L , Kaya A , Kim HY , Manta B , Lee BC , Gladyshev VN . The selenoprotein methionine sulfoxide reductase B1 (MSRB1) . Free Radic Biol Med . 2022 ; 191 : 228 – 40 . doi: 10.1016/J.FREERADBIOMED.2022.08.043 . OpenUrl CrossRef PubMed 39. ↵ Anouar Y , Lihrmann I , Falluel-Morel A , Boukhzar L . Selenoprotein T is a key player in ER proteostasis, endocrine homeostasis and neuroprotection . Free Radic Biol Med . 2018 ; 127 : 145 – 52 . doi: 10.1016/J.FREERADBIOMED.2018.05.076 . OpenUrl CrossRef PubMed 40. ↵ Jeong D won , Kim TS , Chung YW , Lee BJ , Kim IY . Selenoprotein W is a glutathione-dependent antioxidant in vivo . FEBS Lett . 2002 ; 517 : 225 – 8 . doi: 10.1016/S0014-5793(02)02628-5 . OpenUrl CrossRef PubMed Web of Science 41. ↵ Misra S , Lee TJ , Sebastian A , McGuigan J , Liao C , Koo I , et al. Loss of selenoprotein W in murine macrophages alters the hierarchy of selenoprotein expression, redox tone, and mitochondrial functions during inflammation . Redox Biol . 2023 ; 59 . doi: 10.1016/J.REDOX.2022.102571 . OpenUrl CrossRef 42. ↵ Zhang X , Xiong W , Chen LL , Huang JQ , Lei XG . Selenoprotein V protects against endoplasmic reticulum stress and oxidative injury induced by pro-oxidants . Free Radic Biol Med . 2020 ; 160 : 670 – 9 . doi: 10.1016/J.FREERADBIOMED.2020.08.011 . OpenUrl CrossRef PubMed 43. ↵ Dikiy A , Novoselov S V ., Fomenko DE , Sengupta A , Carlson BA , Cerny RL , et al. SelT, SelW, SelH, and Rdx12: Genomics and Molecular Insights into the Functions of Selenoproteins of a Novel Thioredoxin-like Family† . Biochemistry . 2007 ; 46 : 6871 – 82 . doi: 10.1021/BI602462Q . OpenUrl CrossRef PubMed Web of Science 44. ↵ Baclaocos J , Santesmasses D , Mariotti M , Bierła K , Vetick MB , Lynch S , et al. Processive Recoding and Metazoan Evolution of Selenoprotein P: Up to 132 UGAs in Molluscs . J Mol Biol . 2019 ; 431 : 4381 – 407 . doi: 10.1016/J.JMB.2019.08.007 . OpenUrl CrossRef PubMed 45. ↵ Yim SH , Everley RA , Schildberg FA , Lee SG , Orsi A , Barbati ZR , et al. Role of Selenof as a Gatekeeper of Secreted Disulfide-Rich Glycoproteins . Cell Rep . 2018 ; 23 : 1387 – 98 . doi: 10.1016/J.CELREP.2018.04.009 . OpenUrl CrossRef PubMed 46. ↵ Flowers B , Bochnacka O , Poles A , Diamond AM , Kastrati I . Distinct Roles of SELENOF in Different Human Cancers . Biomolecules . 2023 ; 13 . doi: 10.3390/BIOM13030486 . OpenUrl CrossRef 47. ↵ Zigrossi A , Hong LK , Ekyalongo RC , Cruz-Alvarez C , Gornick E , Diamond AM , et al. SELENOF is a new tumor suppressor in breast cancer . Oncogene 2021 41:9. 2022 ; 41 : 1263 – 8 . doi: 10.1038/s41388-021-02158-w . OpenUrl CrossRef 48. ↵ Bertz M , Kühn K , Koeberle SC , Müller MF , Hoelzer D , Thies K , et al. Selenoprotein H controls cell cycle progression and proliferation of human colorectal cancer cells . Free Radic Biol Med . 2018 ; 127 : 98 – 107 . doi: 10.1016/J.FREERADBIOMED.2018.01.010 . OpenUrl CrossRef PubMed 49. Cox AG , Tsomides A , Kim AJ , Saunders D , Hwang KL , Evason KJ , et al. Selenoprotein H is an essential regulator of redox homeostasis that cooperates with p53 in development and tumorigenesis . Proc Natl Acad Sci U S A . 2016 ; 113 : E5562 – 71 . doi: 10.1073/PNAS.1600204113 . OpenUrl Abstract / FREE Full Text 50. ↵ Wu RTY , Cao L , Chen BPC , Cheng WH . Selenoprotein H suppresses cellular senescence through genome maintenance and redox regulation . J Biol Chem . 2014 ; 289 : 34378 – 88 . doi: 10.1074/JBC.M114.611970 . OpenUrl Abstract / FREE Full Text 51. ↵ Lien S , Koop BF , Sandve SR , Miller JR , Kent MP , Nome T , et al. The Atlantic salmon genome provides insights into rediploidization . Nature 2016 533:7602. 2016 ; 533 : 200 – 5 . doi: 10.1038/nature17164 . OpenUrl CrossRef PubMed 52. ↵ Bianchi D , Borza R , De Zan E , Huelsz-Prince G , Gregoricchio S , Dekker M , et al. Zincore, an atypical coregulator, binds zinc finger transcription factors to control gene expression . Science . 2025 ; 389 . doi: 10.1126/science.adv2861 . OpenUrl CrossRef 53. ↵ Odunsi A , Kapitonova MA , Woodward G , Rahmani E , Ghelichkhani F , Liu J , et al. Selenoprotein K at the intersection of cellular pathways . Arch Biochem Biophys . 2025 ; 764 . doi: 10.1016/J.ABB.2024.110221 . OpenUrl CrossRef 54. ↵ Shi Z , Han Z , Chen J , Zhou JC . Endoplasmic reticulum-resident selenoproteins and their roles in glucose and lipid metabolic disorders . Biochim Biophys Acta Mol Basis Dis . 2024 ; 1870 . doi: 10.1016/J.BBADIS.2024.167246 . OpenUrl CrossRef 55. ↵ Chernorudskiy A , Varone E , Colombo SF , Fumagalli S , Cagnotto A , Cattaneo A , et al. Selenoprotein N is an endoplasmic reticulum calcium sensor that links luminal calcium levels to a redox activity . Proc Natl Acad Sci U S A . 2020 ; 117 : 21288 – 98 . doi: 10.1073/PNAS.2003847117/-/DCSUPPLEMENTAL . OpenUrl Abstract / FREE Full Text 56. ↵ Lobanov A V ., Fomenko DE , Zhang Y , Sengupta A , Hatfield DL , Gladyshev VN . Evolutionary dynamics of eukaryotic selenoproteomes: Large selenoproteomes may associate with aquatic life and small with terrestrial life . Genome Biol . 2007 ; 8 : 1 – 16 . doi: 10.1186/GB-2007-8-9-R198/FIGURES/7 . OpenUrl CrossRef PubMed 57. ↵ Sarangi GK , Romagne F , Castellano S . Distinct Patterns of Selection in Selenium-Dependent Genes between Land and Aquatic Vertebrates . Mol Biol Evol . 2018 ; 35 : 1744 – 56 . doi: 10.1093/MOLBEV/MSY070 . OpenUrl CrossRef PubMed 58. ↵ Maroney MJ , Hondal RJ . Selenium versus sulfur: Reversibility of chemical reactions and resistance to permanent oxidation in proteins and nucleic acids . Free Radic Biol Med . 2018 ; 127 : 228 – 37 . doi: 10.1016/J.FREERADBIOMED.2018.03.035 . OpenUrl CrossRef PubMed 59. ↵ Reich HJ , Hondal RJ . Why Nature Chose Selenium . ACS Chem Biol . 2016 ; 11 : 821 – 41 . doi: 10.1021/ACSCHEMBIO.6B00031 . OpenUrl CrossRef PubMed 60. ↵ Castellano S , Andrés AM , Bosch E , Bayes M , Guigó R , Clark AG . Low exchangeability of selenocysteine, the 21st amino acid, in vertebrate proteins . Mol Biol Evol . 2009 ; 26 : 2031 – 40 . doi: 10.1093/MOLBEV/MSP109 . OpenUrl CrossRef PubMed Web of Science 61. ↵ Kaessmann H . Origins, evolution, and phenotypic impact of new genes . Genome Res . 2010 ; 20 : 1313 – 26 . doi: 10.1101/GR.101386.109 . OpenUrl Abstract / FREE Full Text 62. ↵ Santesmasses D , Mariotti M , Gladyshev VN . Bioinformatics of Selenoproteins . Antioxid Redox Signal . 2020 ; 33 : 525 – 36 . doi: 10.1089/ARS.2020.8044 . OpenUrl CrossRef PubMed 63. ↵ Hughes LC , Ortí G , Huang Y , Sun Y , Baldwin CC , Thompson AW , et al. Comprehensive phylogeny of ray-finned fishes (Actinopterygii) based on transcriptomic and genomic data . Proc Natl Acad Sci U S A . 2018 ; 115 : 6249 – 54 . doi: 10.1073/PNAS.1719358115/SUPPL_FILE/PNAS.1719358115.SAPP.PDF . OpenUrl Abstract / FREE Full Text 64. ↵ Cunningham F , Allen JE , Allen J , Alvarez-Jarreta J , Amode MR , Armean IM , et al. Ensembl 2022 . Nucleic Acids Res . 2022 ; 50 : D988 – 95 . doi: 10.1093/NAR/GKAB1049 . OpenUrl CrossRef PubMed 65. ↵ Mariotti M , Guigó R . Selenoprofiles: profile-based scanning of eukaryotic genome sequences for selenoprotein genes . Bioinformatics . 2010 ; 26 : 2656 – 63 . doi: 10.1093/BIOINFORMATICS/BTQ516 . OpenUrl CrossRef PubMed Web of Science 66. ↵ Santesmasses D , Mariotti M , Guigó R . Selenoprofiles: A Computational Pipeline for Annotation of Selenoproteins . Methods in Molecular Biology . 2018 ; 1661 : 17 – 28 . doi: 10.1007/978-1-4939-7258-6_2 . OpenUrl CrossRef PubMed 67. ↵ Sievers F , Higgins DG . Clustal Omega for making accurate alignments of many protein sequences . Protein Sci . 2018 ; 27 : 135 – 45 . doi: 10.1002/pro.3290 . OpenUrl CrossRef PubMed 68. ↵ Wong TKF , Ly-Trong N , Ren H , Baños H , Roger AJ , Susko E , et al. IQ-TREE 3: Phylogenomic Inference Software using Complex Evolutionary Models . 2025 . doi: 10.32942/X2P62N . OpenUrl CrossRef 69. ↵ Huerta-Cepas J , Serra F , Bork P . ETE 3: Reconstruction, Analysis, and Visualization of Phylogenomic Data . Mol Biol Evol . 2016 ; 33 : 1635 . doi: 10.1093/MOLBEV/MSW046 . OpenUrl CrossRef PubMed 70. ↵ Camacho C , Coulouris G , Avagyan V , Ma N , Papadopoulos J , Bealer K , et al. BLAST+: architecture and applications . BMC Bioinformatics . 2009 ; 10 . doi: 10.1186/1471-2105-10-421 . OpenUrl CrossRef PubMed 71. ↵ Jones P , Binns D , Chang HY , Fraser M , Li W , McAnulla C , et al. InterProScan 5: genome-scale protein function classification . Bioinformatics . 2014 ; 30 : 1236 – 40 . doi: 10.1093/BIOINFORMATICS/BTU031 . OpenUrl CrossRef PubMed Web of Science 72. ↵ Stovner EB , Ticó M , Muñoz Del Campo E , Pallarès-Albanell J , Chawla K , Saetrom P , et al. Pyranges v1: a Python framework for ultrafast sequence interval operations . bioRxiv . 2025 ;:2025.12.11.693639. doi: 10.64898/2025.12.11.693639 . OpenUrl Abstract / FREE Full Text 73. ↵ Regina BF , Gladyshev VN , Arnér ES , Berry MJ , Bruford EA , Burk RF , et al. Selenoprotein gene nomenclature . Journal of Biological Chemistry . 2016 ; 291 : 24036 – 40 . doi: 10.1074/jbc.M116.756155 . OpenUrl Abstract / FREE Full Text 74. ↵ Yu D , Ren Y , Uesaka M , Beavan AJS , Muffato M , Shen J , et al. Hagfish genome elucidates vertebrate whole-genome duplication events and their evolutionary consequences . Nat Ecol Evol . 2024 ; 8 : 519 – 35 . doi: 10.1038/s41559-023-02299-z . OpenUrl CrossRef 75. Nakatani Y , Shingate P , Ravi V , Pillai NE , Prasad A , McLysaght A , et al. Reconstruction of proto-vertebrate, proto-cyclostome and proto-gnathostome genomes provides new insights into early vertebrate evolution . Nature Communications 2021 12:1. 2021 ; 12 : 4489 -. doi: 10.1038/s41467-021-24573-z . OpenUrl CrossRef PubMed 76. Dehal P , Boore JL . Two Rounds of Whole Genome Duplication in the Ancestral Vertebrate . PLoS Biol . 2005 ; 3 : e314 . doi: 10.1371/journal.pbio.0030314 . OpenUrl CrossRef PubMed 77. Qi M , Clark J , Moody ERR , Pisani D , Donoghue PCJ . Molecular Dating of the Teleost Whole Genome Duplication (3R) Is Compatible With the Expectations of Delayed Rediploidization . Genome Biol Evol . 2024 ; 16 . doi: 10.1093/gbe/evae128 . OpenUrl CrossRef PubMed 78. Macqueen DJ , Johnston IA . A well-constrained estimate for the timing of the salmonid whole genome duplication reveals major decoupling from species diversification . Proceedings of the Royal Society B: Biological Sciences . 2014 ; 281 : 20132881 . doi: 10.1098/rspb.2013.2881 . OpenUrl CrossRef PubMed 79. ↵ Xu P , Xu J , Liu G , Chen L , Zhou Z , Peng W , et al. The allotetraploid origin and asymmetrical genome evolution of the common carp Cyprinus carpio . Nature Communications 2019 10:1. 2019 ; 10 : 4625 -. doi: 10.1038/s41467-019-12644-1 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted March 13, 2026. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Tracing the vertebrate selenoproteome evolution reveals expansions in ray-finned fishes and convergent depletions in tetrapods Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Tracing the vertebrate selenoproteome evolution reveals expansions in ray-finned fishes and convergent depletions in tetrapods Max Ticó , Jesus Lozano-Fernandez , Marco Mariotti bioRxiv 2025.05.29.656587; doi: https://doi.org/10.1101/2025.05.29.656587 Share This Article: Copy Citation Tools Tracing the vertebrate selenoproteome evolution reveals expansions in ray-finned fishes and convergent depletions in tetrapods Max Ticó , Jesus Lozano-Fernandez , Marco Mariotti bioRxiv 2025.05.29.656587; doi: https://doi.org/10.1101/2025.05.29.656587 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Evolutionary Biology Subject Areas All Articles Animal Behavior and Cognition (7618) Biochemistry (17633) Bioengineering (13857) Bioinformatics (41841) Biophysics (21399) Cancer Biology (18529) Cell Biology (25422) Clinical Trials (138) Developmental Biology (13352) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24282) Genetics (15582) Genomics (22462) Immunology (17700) Microbiology (40295) Molecular Biology (17140) Neuroscience (88421) Paleontology (666) Pathology (2823) Pharmacology and Toxicology (4813) Physiology (7632) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4284) Systems Biology (9808) Zoology (2267)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.