SoyOmics: A deeply integrated database on soybean multi-omics

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

As one of the most important crops to supply majority plant oil and protein for the whole world, soybean is facing an increasing global demand. Up to now, vast multi-omics data of soybean were generated, thereby providing valuable resources for functional study and molecular breeding. Nevertheless, it is tremendously challenging for researchers to deal with these big multi-omics data, particularly considering the unprecedented rate of data growth. Therefore, we collect the reported high-quality omics, including assembly genomes, graph pan-genome, resequencing and phenotypic data of representative germplasms, transcriptomic and epigenomic data from different tissues, organs and accessions, and construct an integrated soybean multi-omics database, named SoyOmics ( https://ngdc.cncb.ac.cn/soyomics ). By equipping with multiple analysis modules and toolkits, SoyOmics is of great utility to facilitate the global scientific community to fully use these big omics datasets for a wide range of soybean studies from fundamental functional investigation to molecular breeding.
Full text 36,119 characters · extracted from preprint-html · click to expand
SoyOmics: A deeply integrated database on soybean multi-omics | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results SoyOmics: A deeply integrated database on soybean multi-omics Yucheng Liu , Yang Zhang , Xiaonan Liu , Yanting Shen , Dongmei Tian , Xiaoyue Yang , Shulin Liu , Lingbin Ni , Zhang Zhang , Shuhui Song , Zhixi Tian doi: https://doi.org/10.1101/2023.02.05.527102 Yucheng Liu 1 State Key Laboratory of Plant Cell and Chromosome Engineering, Institute of Genetics and Developmental Biology, Chinese Academy of Sciences , Beijing 100101, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yang Zhang 2 National Genomics Data Center & CAS Key Laboratory of Genome Sciences and Information, Beijing Institute of Genomics, Chinese Academy of Sciences , Beijing 100101, China 3 China National Center for Bioinformation , Beijing 100101, China 4 University of Chinese Academy of Sciences , Beijing 100039, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xiaonan Liu 2 National Genomics Data Center & CAS Key Laboratory of Genome Sciences and Information, Beijing Institute of Genomics, Chinese Academy of Sciences , Beijing 100101, China 3 China National Center for Bioinformation , Beijing 100101, China 4 University of Chinese Academy of Sciences , Beijing 100039, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yanting Shen 1 State Key Laboratory of Plant Cell and Chromosome Engineering, Institute of Genetics and Developmental Biology, Chinese Academy of Sciences , Beijing 100101, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Dongmei Tian 2 National Genomics Data Center & CAS Key Laboratory of Genome Sciences and Information, Beijing Institute of Genomics, Chinese Academy of Sciences , Beijing 100101, China 3 China National Center for Bioinformation , Beijing 100101, China 4 University of Chinese Academy of Sciences , Beijing 100039, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xiaoyue Yang 1 State Key Laboratory of Plant Cell and Chromosome Engineering, Institute of Genetics and Developmental Biology, Chinese Academy of Sciences , Beijing 100101, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Shulin Liu 1 State Key Laboratory of Plant Cell and Chromosome Engineering, Institute of Genetics and Developmental Biology, Chinese Academy of Sciences , Beijing 100101, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lingbin Ni 1 State Key Laboratory of Plant Cell and Chromosome Engineering, Institute of Genetics and Developmental Biology, Chinese Academy of Sciences , Beijing 100101, China 4 University of Chinese Academy of Sciences , Beijing 100039, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Zhang Zhang 2 National Genomics Data Center & CAS Key Laboratory of Genome Sciences and Information, Beijing Institute of Genomics, Chinese Academy of Sciences , Beijing 100101, China 3 China National Center for Bioinformation , Beijing 100101, China 4 University of Chinese Academy of Sciences , Beijing 100039, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: zhangzhang{at}big.ac.cn songshh{at}big.ac.cn zxtian{at}genetics.ac.cn Shuhui Song 2 National Genomics Data Center & CAS Key Laboratory of Genome Sciences and Information, Beijing Institute of Genomics, Chinese Academy of Sciences , Beijing 100101, China 3 China National Center for Bioinformation , Beijing 100101, China 4 University of Chinese Academy of Sciences , Beijing 100039, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: zhangzhang{at}big.ac.cn songshh{at}big.ac.cn zxtian{at}genetics.ac.cn Zhixi Tian 1 State Key Laboratory of Plant Cell and Chromosome Engineering, Institute of Genetics and Developmental Biology, Chinese Academy of Sciences , Beijing 100101, China 4 University of Chinese Academy of Sciences , Beijing 100039, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: zhangzhang{at}big.ac.cn songshh{at}big.ac.cn zxtian{at}genetics.ac.cn Abstract Full Text Info/History Metrics Preview PDF Abstract As one of the most important crops to supply majority plant oil and protein for the whole world, soybean is facing an increasing global demand. Up to now, vast multi-omics data of soybean were generated, thereby providing valuable resources for functional study and molecular breeding. Nevertheless, it is tremendously challenging for researchers to deal with these big multi-omics data, particularly considering the unprecedented rate of data growth. Therefore, we collect the reported high-quality omics, including assembly genomes, graph pan-genome, resequencing and phenotypic data of representative germplasms, transcriptomic and epigenomic data from different tissues, organs and accessions, and construct an integrated soybean multi-omics database, named SoyOmics ( https://ngdc.cncb.ac.cn/soyomics ). By equipping with multiple analysis modules and toolkits, SoyOmics is of great utility to facilitate the global scientific community to fully use these big omics datasets for a wide range of soybean studies from fundamental functional investigation to molecular breeding. Dear Editor As one of the most important crops to supply majority plant oil and protein for the whole world, soybean is facing an increasing global demand ( Ray et al., 2013 ). The reference genome of accession “Williams82” opened the gate of genomics research in soybean ( Schmutz et al., 2010 ). After that, vast multi-omics data were generated, thereby providing valuable resources for functional study and molecular breeding ( Zhang et al., 2022 ). Nevertheless, it is tremendously challenging for researchers to deal with these big multi-omics data, particularly considering the unprecedented rate of data growth ( Yang et al., 2021 ; Cao et al., 2022 ). Thus, constructing an integrated multiomics database for soybean that provides a one-stop solution for big data mining with friendly interactivity is highly desired. Here, we collect the reported high-quality omics data ( Figure 1A ), including assembly genomes, graph pan-genome, resequencing and phenotypic data of representative germplasms ( Zhou et al., 2015 ; Fang et al., 2017 ; Liu et al., 2020 ), transcriptomic ( Shen et al., 2014 ; Shen et al., 2019 ) and epigenomic ( Shen et al., 2018 ) data from different tissues, organs and accessions, and construct an integrated soybean multi-omics database, named SoyOmics ( https://ngdc.cncb.ac.cn/soyomics ). By equipping with multiple analysis modules and toolkits, SoyOmics is of great utility to facilitate the global scientific community to fully use these big omics datasets for a wide range of soybean studies from fundamental functional investigation to molecular breeding. Download figure Open in new tab Figure 1. Overview of SoyOmics and its practice for data mining. (A) Framework of SoyOmics, including data resource, module organization and application scenarios. (B) Germplasms grouped by seed coat color. (C) GWAS analysis with seed coat color. Threshold is set by P = 1e-10. (D) Gene structure and variation list of SoyZH13_01G182000 . (E) Seed coat color grouped by genotypes of soy1873072. Multiple comparison is conducted by Student-Newman-Keuls (SNK) test, with significant level equal to 0.01. (F) Homologous gene number of SoyZH13_01G182000 in 29 soybean genomes. (G) Expression of SoyZH13_01G182000 in 28 tissues. (H) Change consequence on mRNA and present knowledge of variation soy1873072. (I) Linkage disequilibrium heat map around soy1873072. The left chart shows heat map of all variation pairs in the block, and the right shows heat map of variation pairs with r2 ≥ 0.9. (J) Genotype frequency distribution of G gene in Soja , landrace and cultivar. (K) Selective test signal of genomic region Chr01:54.5-57.4 Mb. Green dash line means the significant threshold. Blue triangle shows the location of SoyZH13_01G182000 . An overview of SoyOmics By integrating different multi-omics data, we develop six highly interactive basic modules in SoyOmics: Genome, Variome, Transcriptome, Phenome, Homology, and Synteny ( Figure 1A ). The Genome module embodies the information of 2,898 soybean germplasms and 27 de novo assembled genomes, providing users with open access to basic information of sequenced germplasms, individual genomes and genes ( Supplemental Figure 1 ). The Variome module organizes approximately 38 million SNPs and INDELs of the 2,898 soybean accessions, facilitating users to check the variation information and whole genome selective signals for any germplasm of interest ( Supplemental Figure 2 ). The Transcriptome module contains two datasets of gene expression: one is from 27 tissues at different developmental stages from Williams82 and ZH13 accessions, respectively, and the other is from 9 tissues at different developmental stages from each of the 26 accessions used for pan-genome analysis. In this module, users can obtain gene expression profiles and gene orthologous information by specifying gene ID or functional description ( Supplemental Figure 3 ). The Phenome module collects approximately 27 thousand records of 115 phenotypes with terms defined as controlled vocabularies that fall into 5 classes (including morphology, growth and development, biochemistry, biotic stress, and vigor) as well as 17 subclasses ( Supplemental Figure 4 ). The Homology module displays the soybean pan-genome by characterizing 57,480 homologous gene groups. Users can specify any gene ID, homologous group ID or gene functional description to retrieve the homologous group of interest ( Supplemental Figure 5 ). The Synteny module deposits approximately 550 thousand large-scale structural variations (SVs) in the pan-genome, in which users can visualize and download the SVs and synteny blocks by setting a specific genomic region. Furthermore, the graph pan-genome is embedded and a SequenceTubeMap web service ( https://github.com/vgteam/sequenceTubeMap ) is deployed for visualization of pan-genome threads (or haplotypes) according to nodes made up by SVs ( Supplemental Figure 6 ). Download figure Open in new tab Supplemental Figure 1. Introduction of Genome module. (A) Result of searching by germplasm. User can redirect to Phenome module and SoyArray tool from germplasm private page. (B) Result of searching by genome. Users can redirect to gene search panel and Synteny module from the assembly private page. (C) Result of searching by gene. Search panel affords setting of genome, gene type, genomic position and gene ID. Users can redirect to Variome, Homology and Transcriptome module from the gene private page. Download figure Open in new tab Supplemental Figure 2. Introduction of Variome module. (A) Result of searching by variation. Search panel affords setting of genomic position, variation type, consequence on sequence, present knowledge and frequency. Users can redirect to gene private page and Phenome modules from variation private pate. (B) Result of searching by domestication signature. Track can be select/unselect to modify the visualization result. Download figure Open in new tab Supplemental Figure 3. Introduction of Transcriptome module. This module shows expression of gene and their orthologous among tissues. Users can redirect to the gene private page for here. Expression value table can be hidden/appeared from here. Download figure Open in new tab Supplemental Figure 4. Introduction of Phenome module. This module shows introduction, records and statistics of phenotypes. Users can redirect to the germplasm private page from here. Download figure Open in new tab Supplemental Figure 5. Introduction of Homology module. This module shows a homologous gene family with gene members among soybean genomes and the integrated annotation. Users can redirect to the gene private page here and download the CDS/protein sequences and phylogeny tree. Download figure Open in new tab Supplemental Figure 6. Introduction of Synteny module. (A) Chromosome synteny visualization and list of SVs between ZH13 and other soybean genomes. Users can redirect to any de novo assembled genome from here. (B) Graph genome visualization and haplotype identification according to nodes of graph. In addition, SoyOmics is designed to provide user-friendly search bar in each module and to cover as much as more possible substances. According to the searching category and inputting context, it features powerful search engine to provide comprehensive associated results with friendly links from one module to other modules ( Supplemental Figure 7 ). Download figure Open in new tab Supplemental Figure 7. The comprehensive logic of searching bar. The global search provides all related result of the searched item in the SoyOmics. The directional search exactly jumps to the private page of the searched entity. Application toolkits In addition to the six modules, we design several commonly easy-to-use toolkits, including easyGWAS, ExpPattern, HapSnap, BLAST, VersionMap , and SoyArray ( Figure 1A ). The easyGWAS is a tool for quick-start genome-wide association study (GWAS) analysis, providing friendly interface for parameters setting and algorithms selection and offering multiple high-quality outputs including Manhattan plot, QQ-plot and text result ( Supplemental Figure 8 ). The ExpPattern is for conducting expression pattern analysis for a gene list against soybean tissues. It can generate expression heatmap, with options of whether to execute clustering or not ( Supplemental Figure 9 ). Besides, the tspex is incorporated in the ExpPattern for advice of gene’s tissue-specificity ( Camargo et al., 2020 ). The HapSnap is designed for haplotype analysis for a genomic region. Users can refine the variations via selection of variation type and quality control. The output includes haplotype frequency, haplotype vs. genotype, and linkage disequilibrium ( Supplemental Figure 10 ). The VersionMap is capable to convert the genomic region between ZH13 (v2) and other de novo genomes of soybean, or gene ID between Williams82 (v2) and ZH13 (v2) ( Supplemental Figure 11 ). Download figure Open in new tab Supplemental Figure 8. Guidance of easyGWAS with a case study of leaf width. Download figure Open in new tab Supplemental Figure 9. Guidance of ExpPattern with a case study of MADS-box family. Download figure Open in new tab Supplemental Figure 10. Guidance of HapSnap with a case study of GmZEP . Download figure Open in new tab Supplemental Figure 11. Guidance of VersionMap with examples of genomic regions and gene IDs. We also develop a toolkit named SoyArray by embedding the information of GenoBaits soybean array ( Liu et al., 2022 ), in which users can search and download the marker information they are interested in. We also afford a function in the SoyArray to compare divergent sites between two germplasms based the makers from GenoBaits soybean array, which is helpful for parents’ picking in genetic or breeding study ( Supplemental Figure 12 ). Download figure Open in new tab Supplemental Figure 12. Guidance of SoyArray with comparison of two soybean accessions. Data mining using SoyOmics As SoyOmics integrates a wide variety of soybean multi-omics data, it can be used for deep mining ranging from fundamental research to molecular breeding. Here we take a previously reported seed coat color causal gene, G ( Wang et al., 2018 ), as an example. In SoyOmics, we can group germplasms by green or yellow seed coat colors ( Figure 1B ). According to the phenotype data, we can easily conduct GWAS using the easyGWAS toolkit, and then identify a significant association signal that is located in the G gene, SoyZH13_01G182000 ( Figure 1C and 1D ). According to the interested association genetic variant, users can get phenotype variations among different genotypes, such as the seed coat color ( Figure 1E ). By searching the candidate gene SoyZH13_01G182000 from different modules, users can obtain a wealth of gene information including basic summary, functional annotation, homology in 29 soybean genomes and expression pattern in 28 tissues ( Figure 1D, 1F , and 1G ). Furthermore, users can also investigate functional annotations for any variant of interest ( Figure 1H ), linkage disequilibrium around the association genetic variant ( Figure 1I ), allele frequency in different populations ( Figure 1J ), and selection sweeps for the association regions by three different test methods ( Figure 1K ). Notably, the majority of charts generated in SoyOmics can be directly downloaded and edited. In summary, SoyOmics features comprehensive integration of multi-omics datasets and provides user-friendly interfaces for soybean study. Undoubtedly, soybean omics data are generated at increasing scales and rates, including resequencing data for more germplasms, transcriptome data from bulk, single-cell and spatial RNA-seq, epigenetic data from Hi-C, ATAC-seq or histone modification, etc. Therefore, future directions for SoyOmics mainly focus on continuous integration of these newly-generated omics data. In addition, artificial intelligence (AI)-based approaches for deep mining of these big data would provide valuable insights for a wide range of soybean studies, particularly for AI breeding in the era of big data. Towards this end, we would like to call for global collaborations to build SoyOmics as a valuable platform for the whole research community around the world. SUPPLEMENTAL INFORMATION Supplemental information is available at Molecular Plant Online. FUNDING This work was supported by the Strategic Priority Research Program of the Chinese Academy of Sciences (XDA24000000, XDA19050302, and XDA24040201), the Science and Technology Innovation 2030 – Major Project (2022ZD04017), the National Natural Science Foundation of China (31788103, 32030021, and 32000475), the National Key Research and Development Program of China (2021YFF1001201), Taishan Scholars Program, and Xplorer Prize Award, the Youth Innovation Promotion Association of the Chinese Academy of Sciences (Y2021038). AUTHOR CONTRIBUTIONS Z.T., and Z.Z. conceived this project. Z.T., S.S., and Z.Z. supervised this work. Y.L., Y.S., X.L., and Y.Z. designed the framework of database and wrote the pipelines. S.L., and X.Y. revised germplasm information and phenotype records. X.L. and Y.Z. constructed the database. L.N. developed the pipeline used in VersionMap module. D.T. built up the easyGWAS, VersionMap and graph-based genome module. Y.L., Y.Z., X.N., S.Y., and Z.T. wrote the manuscript. Z.T., S.S., and Z.Z. revised the manuscript. All authors read and approved the final manuscript. Materials and methods Gene and gene family functional annotation The 27 de novo assembly soybean genomes were annotated by Pfam, InterPro, UniProt, and GO functional knowledgebases. We used InterProScan (version 5.52-86.0) ( Jones et al., 2014 ) to annotate the protein sequences with Pfam and InterPro items. Then BLASTP results of protein sequence against the UniProt Swiss-Prot data were used for UniProt item annotation. Items matching the highest BLAST score, with identity ≥ 30% and length match portion ≥ 50% were treated as the UniProt annotation of genes. Furthermore, we used SANPANZ3, a stand-alone version of PANNZER2 ( Toronen et al., 2018 ), to annotate proteins with GO item and functional description, by ppv over 0.4 as filtration. The KEGG pathway annotation was done with KofamScan ( Aramaki et al., 2020 ). To annotate the function of a gene family, we considered all the functional annotation items of its gene members and built an integrated method for gene group annotation. Suppose that one functional item is annotated to an n -sample homologous group with m (1 ≤ m ≤ n ) samples, we define that N i is the total number of genes for any given sample i ( i = 1, 2, 3… n ) in the homologous group and X i is the number of genes for the sample i that are annotated by the item. Thus, the annotation score ( S ) was formulated as: where α is the weight parameter for balance of the two measure aspects. In our study, α was set at 0.75. Finally, for each type of annotation base, such as Pfam, GO, UniProt, the item with the highest S value was picked as the annotation of the group. For functional description of a homologous group, if any description text is the largest description and 1) is the unique; or 2) possesses over half of the genes and is 1.2 fold larger than the second one; or 3) possesses over thirty percent genes and is 2 fold larger than the second one, we determined the description text as the representative for the group. Modification of phenotype data In SoyOmics, we collected 115 phenotypes, with 18 qualitative traits and 97 quantitative traits. To estimate the relationship between genotype and phenotype, firstly, we coded the 18 qualitative traits by numeric values (Supplemental Table 1). After that, 81 phenotypes collected from multi-areas or -years were modified by best linear unbiased prediction (BLUP) with R package “lme4”. Data passed through this modification were used for genotype-to-phenotype analysis. Selective test Selective test was performed with the SNPs from the 2,898 soybean accessions. Tajima’s D ( Tajima, 1989 ), reduce of diversity (ROD) ( Xu et al., 2011 ), and F ST ( Weir and Cockerham, 1984 ) were calculated with 20 kb bin without overlapping. For Tajima’s D, the bottom 5% value was used as the significant threshold. For ROD and F ST , the top 5% value was chosen as the significant threshold. Variation imputation To get the phased and no gap variation, we conducted imputation for the SNP and INDEL data of 2,898 soybeans. We split variants into the 5 Mb sections and used Beagle (v4.0) ( Browning and Browning, 2007 ) to do the imputation, with parameters of phase-its=3 and impute-its=3. Then the separated variations were combined together by the chromosome and position rank. Variation effects estimation In SoyOmics, we provided the Sorting Intolerant From Tolerant (SIFT) ( Ng and Henikoff, 2001 ) values for SNPs to estimate their potential effects on the changing of amino acid. To do this analysis, the protein sequences from RefSeq and Swiss-Prot were used as target database. The SNP tagged by ‘non-synonymous SNV’, ‘stop-gain’ and ‘stop-loss’ were extracted as query variations. The results with Median Info within 2.75∼3.25 were treated as confident results and shown in SoyOmcis. If SIFT score was over 0.05, the substitution was tagged by Tolerated; otherwise, it was tagged by Damaging. Identification of MADS-box Previous study identified a total of 157 MADS-box members in soybean ( Gramzow and Theissen, 2013 ; Shu et al., 2013 ), with 131 having isoforms in the ZH13 genome. Pfam screening of ZH13 genome identified 184 genes that contained SRF-TF (PF00319) and/or K-box (PF01486) domain, which were the characteristic domains of MADS-box . Of the 184 genes, 130 were same as the reported genes. The genes with PF00319 or PF01486 was used for functional display as an example of ExpPattern . Generate chain file for genome position conversion In SoyOmics, we provided the Genome position function between ZH13 and other soybean genomes. The chain files were generated with the UCSC liftOver pipeline ( Kuhn et al., 2013 ). For each genome with ZH13, we generated two chain files by ZH13 as query or target. Therefore, users can do the bi-directional conversion between any genome and ZH13. The position conversion was done by CrossMap ( Zhao et al., 2014 ) in SoyOmics. View this table: View inline View popup Table S1. Encoding of qualitative phenotypes. ACKNOWLEDGMENTS We thank a number of SoyOmics users for their kind advice. No conflict of interest is declared. Reference ↵ Camargo , A.P. , Vasconcelos , A.A. , Fiamenghi , M.B. , Pereira , G.A.G. , and Carazzolle , M.F. ( 2020 ). tspex: a tissue-specificity calculator for gene expression data . ↵ Cao , P. , Zhao , Y. , Wu , F. , Xin , D. , Liu , C. , Wu , X. , Lv , J. , Chen , Q. , and Qi , Z. ( 2022 ). Multi-Omics Techniques for Soybean Molecular Breeding . Int J Mol Sci 23 . ↵ Fang , C. , Ma , Y. , Wu , S. , Liu , Z. , Wang , Z. , Yang , R. , Hu , G. , Zhou , Z. , Yu , H. , Zhang , M. , et al. ( 2017 ). Genome-wide association studies dissect the genetic networks underlying agronomical traits in soybean . Genome Biol . 18 : 1 – 14 . OpenUrl CrossRef ↵ Liu , Y. , Du , H. , Li , P. , Shen , Y. , Peng , H. , Liu , S. , Zhou , G.-A. , Zhang , H. , Liu , Z. , and Shi , M. ( 2020 ). Pan-genome of wild and cultivated soybeans . Cell 182 , 162 - 176.e113 . OpenUrl CrossRef ↵ Liu , Y. , Liu , S. , Zhang , Z. , Ni , L. , Chen , X. , Ge , Y. , Zhou , G. , and Tian , Z. ( 2022 ). GenoBaits Soy40K: a highly flexible and low-cost SNP array for soybean studies . Sci. China Life Sci . 65 : 1898 – 1901 . OpenUrl ↵ Ray , D.K. , Mueller , N.D. , West , P.C. , and Foley , J.A. ( 2013 ). Yield trends are insufficient to double global crop production by 2050 . PLoS One 8 : e66428 . OpenUrl CrossRef PubMed ↵ Schmutz , J. , Cannon , S.B. , Schlueter , J. , Ma , J. , Mitros , T. , Nelson , W. , Hyten , D.L. , Song , Q. , Thelen , J.J. , Cheng , J. , et al. ( 2010 ). Genome sequence of the palaeopolyploid soybean . Nature 463 : 178 – 183 . OpenUrl CrossRef PubMed Web of Science ↵ Shen , Y. , Du , H. , Liu , Y. , Ni , L. , Wang , Z. , Liang , C. , and Tian , Z. ( 2019 ). Update soybean Zhonghuang 13 genome to a golden reference . Sci. China Life Sci . 62 : 1257 – 1260 . OpenUrl ↵ Shen , Y. , Zhang , J. , Liu , Y. , Liu , S. , Liu , Z. , Duan , Z. , Wang , Z. , Zhu , B. , Guo , Y.-L. , and Tian , Z. ( 2018 ). DNA methylation footprints during soybean domestication and improvement . Genome Biol . 19 : 128 . OpenUrl ↵ Shen , Y. , Zhou , Z. , Wang , Z. , Li , W. , Fang , C. , Wu , M. , Ma , Y. , Liu , T. , Kong , L.A. , Peng , D.L. , et al. ( 2014 ). Global dissection of alternative splicing in paleopolyploid soybean . Plant Cell 26 : 996 – 1008 . OpenUrl Abstract / FREE Full Text ↵ Wang , M. , Li , W. , Fang , C. , Xu , F. , Liu , Y. , Wang , Z. , Yang , R. , Zhang , M. , Liu , S. , Lu , S. , et al. ( 2018 ). Parallel selection on a dormancy gene during domestication of crops from multiple families . Nat Genet . 50 : 1435 – 1441 . OpenUrl CrossRef ↵ Yang , Y. , Saand , M.A. , Huang , L. , Abdelaal , W.B. , Zhang , J. , Wu , Y. , Li , J. , Sirohi , M.H. , and Wang , F. ( 2021 ). Applications of multi-omics technologies for crop improvement . Front Plant Sci . 12 : 563953 . OpenUrl CrossRef ↵ Zhang , M. , Liu , S. , Wang , Z. , Yuan , Y. , Zhang , Z. , Liang , Q. , Yang , X. , Duan , Z. , Liu , Y. , Kong , F. , et al. ( 2022 ). Progress in soybean functional genomics over the past decade . Plant Biotechnol. J . 20 : 256 – 282 . OpenUrl ↵ Zhou , Z. , Jiang , Y. , Wang , Z. , Gou , Z. , Lyu , J. , Li , W. , Yu , Y. , Shu , L. , Zhao , Y. , Ma , Y. , et al. ( 2015 ). Resequencing 302 wild and cultivated accessions identifies genes related to domestication and improvement in soybean . Nat Biotechnol . 33 : 408 – 414 . OpenUrl CrossRef PubMed Reference ↵ Aramaki , T. , Blanc-Mathieu , R. , Endo , H. , Ohkubo , K. , Kanehisa , M. , Goto , S. , and Ogata , H. ( 2020 ). KofamKOALA: KEGG Ortholog assignment based on profile HMM and adaptive score threshold . Bioinformatics 36 : 2251 – 2252 . OpenUrl CrossRef ↵ Browning , S.R. , and Browning , B.L. ( 2007 ). Rapid and accurate haplotype phasing and missing-data inference for whole-genome association studies by use of localized haplotype clustering . Am. J. Hum. Genet . 81 : 1084 – 1097 . OpenUrl CrossRef PubMed Web of Science ↵ Gramzow , L. , and Theissen , G. ( 2013 ). Phylogenomics of MADS-Box genes in plants - two opposing life styles in one gene family . Biology (Basel) 2 : 1150 – 1164 . OpenUrl ↵ Jones , P. , Binns , D. , Chang , H.-Y. , Fraser , M. , Li , W. , McAnulla , C. , McWilliam , H. , Maslen , J. , Mitchell , A. , Nuka , G. , et al. ( 2014 ). InterProScan 5: genome-scale protein function classification . Bioinformatics 30 : 1236 – 1240 . OpenUrl CrossRef PubMed Web of Science ↵ Kuhn , R.M. , Haussler , D. , and Kent , W.J. ( 2013 ). The UCSC genome browser and associated tools . Brief Bioinform . 14 : 144 – 161 . OpenUrl CrossRef PubMed ↵ Ng , P.C. , and Henikoff , S. ( 2001 ). Predicting deleterious amino acid substitutions . Genome Res . 11 : 863 – 874 . OpenUrl Abstract / FREE Full Text ↵ Shu , Y. , Yu , D. , Wang , D. , Guo , D. , and Guo , C. ( 2013 ). Genome-wide survey and expression analysis of the MADS-box gene family in soybean . Mol. Biol. Rep . 40 : 3901 – 3911 . OpenUrl CrossRef PubMed ↵ Tajima , F. ( 1989 ). Statistical method for testing the neutral mutation hypothesis by DNA polymorphism . Genetics 123 : 585 – 595 . OpenUrl Abstract / FREE Full Text ↵ Toronen , P. , Medlar , A. , and Holm , L. ( 2018 ). PANNZER2: a rapid functional annotation web server . Nucleic Acids Res . 46 : W84 – W88 . OpenUrl CrossRef PubMed ↵ Weir , B.S. , and Cockerham , C.C. ( 1984 ). Estimating F-statistics for the analysis of population structure . Evolution 1358 – 1370 . ↵ Xu , X. , Liu , X. , Ge , S. , Jensen , J.D. , Hu , F. , Li , X. , Dong , Y. , Gutenkunst , R.N. , Fang , L. , Huang , L. , et al. ( 2011 ). Resequencing 50 accessions of cultivated and wild rice yields markers for identifying agronomically important genes . Nat Biotechnol . 30 : 105 – 111 . OpenUrl CrossRef PubMed ↵ Zhao , H. , Sun , Z. , Wang , J. , Huang , H. , Kocher , J.P. , and Wang , L. ( 2014 ). CrossMap: a versatile tool for coordinate conversion between genome assemblies . Bioinformatics 30 : 1006 – 1007 . OpenUrl CrossRef PubMed Web of Science Back to top Previous Next Posted February 05, 2023. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following SoyOmics: A deeply integrated database on soybean multi-omics Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share SoyOmics: A deeply integrated database on soybean multi-omics Yucheng Liu , Yang Zhang , Xiaonan Liu , Yanting Shen , Dongmei Tian , Xiaoyue Yang , Shulin Liu , Lingbin Ni , Zhang Zhang , Shuhui Song , Zhixi Tian bioRxiv 2023.02.05.527102; doi: https://doi.org/10.1101/2023.02.05.527102 Share This Article: Copy Citation Tools SoyOmics: A deeply integrated database on soybean multi-omics Yucheng Liu , Yang Zhang , Xiaonan Liu , Yanting Shen , Dongmei Tian , Xiaoyue Yang , Shulin Liu , Lingbin Ni , Zhang Zhang , Shuhui Song , Zhixi Tian bioRxiv 2023.02.05.527102; doi: https://doi.org/10.1101/2023.02.05.527102 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7806) Biochemistry (18223) Bioengineering (14391) Bioinformatics (43122) Biophysics (21998) Cancer Biology (19110) Cell Biology (26193) Clinical Trials (138) Developmental Biology (13651) Ecology (20405) Epidemiology (2067) Evolutionary Biology (24861) Genetics (15859) Genomics (22985) Immunology (18215) Microbiology (41375) Molecular Biology (17552) Neuroscience (90878) Paleontology (679) Pathology (2906) Pharmacology and Toxicology (4955) Physiology (7876) Plant Biology (15518) Scientific Communication and Education (2068) Synthetic Biology (4432) Systems Biology (10014) Zoology (2319) window.__CF$cv$params={r:'a1d5eb01daa78a59',t:'MTc4NDQyNDE3Ng==',u:'019f77f826af7311905b00e3abc5fee5',ut:'_kOjz1sN2DZeM13ghcFE2iOlnR1696AAVKERmSYSFSY-1784424179-1.2.1.1-z4a4AFR.r99TcaMYBSLNIBndPU1moOwT2Xni_Z2gXd1RkPlJ2J_.T8Y0QN1J_Rppwue4rui58g5ECgJVK9DR88wKXVsw9qq9Rrf_PWAAyHM',i:60};(function(){if(!document.body)return;var s=document.createElement('script');s.src='/cdn-cgi/challenge-platform/scripts/precursor/main.js';document.head.appendChild(s);})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. The paper's references may be in our DB but unresolved to ``paper_id`` (resolution happens at ingest when the cited DOI matches a row we already have). Run the cross-source citation reconcile pass to retry.

Source provenance

europepmc
last seen: 2026-05-19T01:45:01.086888+00:00
unpaywall
last seen: 2026-07-20T07:01:09.845843+00:00