Full text
40,824 characters
· extracted from
preprint-html
· click to expand
Genome sequencing, de novo assembly and annotation of the commercially important bamboo, Bambusa tulda Roxb | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Genome sequencing, de novo assembly and annotation of the commercially important bamboo, Bambusa tulda Roxb Sutrisha Kundu , Oliver Rupp , Sonali Dey , Mridushree Basak , Sudeshna Bera , Annette Becker , View ORCID Profile Malay Das doi: https://doi.org/10.1101/2025.07.19.665512 Sutrisha Kundu 1 Plant Genomics Laboratory, Department of Life Sciences, Presidency University, 86/1 College Street , Kolkata-700073, West Bengal, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site Oliver Rupp 2 Bioinformatik und Systembiologie, Justus-Liebig-University , Germany , Gießen, D-35390 Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sonali Dey 1 Plant Genomics Laboratory, Department of Life Sciences, Presidency University, 86/1 College Street , Kolkata-700073, West Bengal, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mridushree Basak 1 Plant Genomics Laboratory, Department of Life Sciences, Presidency University, 86/1 College Street , Kolkata-700073, West Bengal, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sudeshna Bera 1 Plant Genomics Laboratory, Department of Life Sciences, Presidency University, 86/1 College Street , Kolkata-700073, West Bengal, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site Annette Becker 3 Institute of Botany, Justus-Liebig-University , Germany , Gießen, D-35392 Find this author on Google Scholar Find this author on PubMed Search for this author on this site Malay Das 1 Plant Genomics Laboratory, Department of Life Sciences, Presidency University, 86/1 College Street , Kolkata-700073, West Bengal, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Malay Das For correspondence: malay.dbs{at}presiuniv.ac.in Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Bambusa tulda Roxb., a member of the Bambusoideae subfamily, is an ecologically and commercially important plant resource widely distributed on the Indian subcontinent. Our study reports long-read PacBio HiFi sequencing and genome assembly of B. tulda . Flow cytometry analysis revealed its estimated genome size ∼3 Gb. The de novo genome assembly of B. tulda predicted 43 contigs, distributed across three subgenomes, with a total size of 1.37 Gb, contig N50 of 35.69 Mb, and BUSCO score 99%. Repetitive elements constitute 63.31% of the genome. Functional annotation predicted 56,890 protein-coding genes, constituting 19.44% of the genome. This high-quality draft genome assembly will serve as an invaluable resource for future studies on the life history traits, phylogenomic analysis, comparative genomics, and targeted genome modification for important trait improvement of B. tulda . Background and summary Bamboos are among the fastest growing, non-timber, renewable plant resources and are widely distributed throughout the tropics and subtropics [ 1 , 2 ]. They comprise the Bambusoideae subfamily of the monocotyledonous grass family Poaceae, and have significant ecological and economical value [ 3 , 4 ]. The Bambusoideae subfamily is classified into two clades: herbaceous bamboos and woody bamboos [ 5 ]. Herbaceous bamboos possess weakly developed rhizomes, lower lignocellulosic biomass, and exhibit annual or seasonal flowering cycles [ 6 ]. In contrast, the woody bamboos demonstrate rapid culm growth, high accumulation of lignocellulosic biomass, and delayed flowering subsequent upon lengthy vegetative phases upto 120 years [ 1 , 7 – 9 ]. Although, the first bamboo genome Phyllostachys heterocycla (= P . edulis ) was sequenced more than a decade ago [ 10 ], major progress in bamboo genome sequencing was achieved more recently due to the introduction of long read sequencing technologies and advancement in algorithms for genome assembly and annotation [ 11 ]. Twelve species from the Bambusoideae subfamily, with diverse worldwide distribution, have been sequenced and their chromosome-level assemblies have been published ( Fig. 1a ) [ 2 , 5 ]. Download figure Open in new tab Fig. 1. (a) Worldwide native distribution of sequenced bamboo genomes, alongwith B. tulda . The map was drawn based on information obtained from World Flora Online (WFO 2024), Kew (POWO, 2025), and Guadua Bamboo ( https://www.guaduabamboo.com/ ). (b) B. tulda plant in native habitat, (c) leaf, (d) emerging culm, and (e) spikelet inflorescence. Bambusa tulda , also known as the Bengal bamboo, is a paleotropical woody bamboo, native to the Indian subcontinent, and certain parts of southeast Asia ( Fig. 1a, b, c, d , e). It is putatively aneuploid (70-72 chromosomes) [ 12 ] in nature, and possesses highly lignified culms. Its tentative flowering time is ∼50 years [ 13 ] and exhibits both sporadic and gregarious flowering [ 14 ]. In addition to their conventional use as construction wood in rural housing in south and southeast Asia, the plant draws renewed attention due to their potential use in the paper-pulp industry [ 15 ] and bio-energy sector [ 9 ]. In vitro micropropagation method have also been optimized for B. tulda , providing opportunity for genetic intervention [ 16 ]. Until now, some flowering-associated genes were identified for functional studies in B. tulda flowering [ 17 – 20 ]. However, in the absence of a good quality, whole genome sequence, comparative evolutionary studies and genetic manipulation for trait improvement remain challenging. This study aimed to sequence, assemble, and annotate the B. tulda genome to provide a genomic resource for this valuable species. Long-read, PacBio high-fidelity (HiFi) sequencing was performed to obtain a draft genome assembly of B. tulda . The genome size estimated by flow-cytometry and k -mer analyses were ∼3 Gb and ∼1.17 Gb (-k 25, p: 2), respectively ( Table 1 , Fig. 2 ). Approximately 116.4 Gb raw PacBio HiFi data and 76.8 Gb transcriptome data was obtained ( Tables 2 , 3 ). The primary contig assembly, generated through HiFiasm, produced 752 contigs with the contig N50 value 31.02 Mb ( Table 4 ). Two haplotypes were obtained from a haplotype-resolved assembly, with haplotype 1 containing 827 and haplotype 2 containing 326 contigs ( Table 4 ). Further refinement and mapping of the primary assembly with other sequenced bamboo genomes produced a final draft genome assembly of 43 contigs with contig N50 value 35.69 Mb ( Table 5 ). Genome assembly quality confirmed a BUSCO score of ∼99% ( Fig. 3 ). Genome annotation identified 56,890 protein-coding genes, constituting 19.44% of the genome ( Table 6 ). Repetitive sequences constituted major part (63.31%) of the B. tulda genome ( Table 7 ). Additionally, 1,355 rRNA and 1019 tRNA genes were identified ( Table 8 ). This draft genome assembly and annotation of B. tulda will serve as a valuable genetic resource for future studies (Box1). Download figure Open in new tab Fig. 2. Genome size estimation of B . tulda . (a) Flow cytometry based estimation of genome size from fluorescence intensity histograms showing mean peak positions of B. tulda , Solanum lycopersicum , and Zea mays . (b) Genome size estimation based on k -mer distribution (“-k 25”) using Jellyfish-2 and Genomescope 2.0. The analysis also reveals unique sequences, with both homozygous and heterozygous distribution. Download figure Open in new tab Fig. 3. Evaluation of genome assembly completeness of B. tulda against the Embryophyte (n = 425) and Viridiplantae (n = 1614) lineage using BUSCO analysis, demonstrating the complete, fragmented, and missing BUSCO groups. View this table: View inline View popup Download powerpoint Table 1. Estimation of genome size using flow cytometry analyses. View this table: View inline View popup Download powerpoint Table 2. Sequencing statistics of B. tulda genome. View this table: View inline View popup Download powerpoint Table 3. Sequencing statistics of B. tulda transcriptome. View this table: View inline View popup Download powerpoint Table 4. Assembly statistics of the primary and haplotype-resolved B. tulda genome. View this table: View inline View popup Download powerpoint Table 5. Assembly statistics of the final draft genome of B. tulda View this table: View inline View popup Table 6. Statistics of gene prediction in B. tulda genome. View this table: View inline View popup Table 7. Statistics of repeat elements in the B. tulda genome. View this table: View inline View popup Download powerpoint Table 8. Statistics of non-coding rRNA and tRNA prediction in B. tulda genome. Box 1. Important research questions in B. tulda biology What is the polyploidization history of B. tulda ? Why is flowering time delayed up to ∼50 years? What are the major regulators promoting rapid vegetative growth? What is the role of repeat elements in bamboo genome evolution? What are the major gene families that expanded/contracted in the bamboo lineages? Materials and methods Sample collection Tissues were collected from a natural, flowering population of B. tulda , located at Rahuta, Shyamnagar (22.83°N, 88.40°E), West Bengal, India. Fresh young leaves collected were used for genome size estimation and whole genome sequencing, while for transcriptome sequencing, emerging culm was used ( Figs. 1B, C, D ). All the collected tissues were quickly flash-frozen in dry ice, and subsequently stored at -80°C. Genome size estimation by flow cytometry analysis For genome size estimation, two plants with known genome sizes: Solanum lycopersicum L. Stupické polnı’ rané (1.96 pg) and Zea mays L. ‘CE-777’ (5.43 pg) were used as internal references. Isolation of nuclei was carried out by following the nuclei-isolation protocol of Dolezel [ 21 ]. In brief, 40 mg fresh young leaf tissue from both the reference plants and B. tulda were finely chopped in 1 ml ice-cold nuclei isolation Galbraith buffer. The homogenate was filtered through a 40 μm nylon mesh. The filtrate was treated simultaneously with 50 μg/ml RNase A and 50 μg/ml propidium iodide. After incubation in the dark for 30 mins, the suspension of stained nuclei was introduced into flow cytometer (S3e Cell Sorter, Bio-Rad), using argon-ion laser of 488 nm. Live gating was set around the fluorescence area signal (FL3-A) to obtain cell cycle histograms. Based on the fluorescence histograms, the nuclear DNA content was calculated. The estimated genome size of B. tulda was 3.01 Gb and 3.02 Gb on the basis of the two reference plants S. lycopersicum and Z. mays , respectively ( Table 1 ). Genome and transcriptome sequencing High molecular weight (HMW) genomic DNA was extracted from the young leaves using the NucleoBond® HMW DNA kit (Macherey-Nagel, Germany), following the manufacturer’s protocol. The extracted HMW genomic DNA was fragmented to obtain Single-Molecule Real-Time (SMRT) bell libraries. The SMRTbell libraries were then sequenced in the PacBio Sequel IIe (3 libraries) and PacBio Revio (2 libraries) platform (Novogene, Genomics Singapore Pte. Ltd) (Fig. S1, S2). The raw reads obtained by sequencing 5 libraries were pooled together and subjected to quality control. Genome sequencing in the PacBio Sequel IIe platform yielded 36.75 Gb data with N50 value of 12.73 kb and mean read length 12.62 kb ( Table 2 ). Similarly, genome sequencing in the PacBio Revio platform yielded 79.63 Gb data with N50 value of 11.87 kb and mean read length 11.6 kb and ( Table 2 ). From the two sequencing platforms, a total of 10264295 PacBio SMRT reads corresponding to 116.38 Gb data was obtained ( Table 2 ). For accurate annotation of the B. tulda genome, transcriptome sequencing was performed in the Illumina Novaseq 6000 platform. Total RNA was extracted from tissues using the RNeasy® Plant Mini Kit (Qiagen), following the manufacturer’s protocol. The integrity of isolated RNAs was examined in the Agilent 5400 Bioanalyzer, and sample with RNA integrity value >8 were used for library construction. The library was subjected to the Illumina NovaSeq 6000 PE150 platform for sequencing. Transcriptome sequencing yielded a total raw data output of 76.8 Gb with a Q30 value of 93.08% ( Table 3 ). Estimation of genome characteristics Following genome sequencing, k -mer analysis was performed to estimate the genome size, heterozygosity rate, and proportion of repetitive sequences of B. tulda . The distribution of k -mer frequency from the raw PacBio HiFi reads was performed in Jellyfish (v2.3.0) [ 22 ] software. The k -mer spectrum obtained for “-k 25”, ploidy = 2 was then analyzed using the GenomeScope 2.0 [ 23 ] web tool. The predicted genome size was ∼1.17 Gb, heterozygosity rate 3.85%, sequencing error rate 0.143% and repetitive sequences constituted 44.8% of the genome ( Fig. 2 ). Smaller peaks at 3x and 4x coverage indicated possible genome duplication. De novo genome assembly The clean PacBio HiFi reads were assembled de novo into a draft haplotype-resolved assembly and a primary contig assembly (a combination of both haplotypes) using Hifiasm v0.16.1-R341 [ 24 ] tool. The primary contig assembly produced a total of 752 contigs equivalent to 1.85 Gb of data with the N50 value 31.02 Mb ( Table 4 , Fig. S3a). Size of the two haplotype assemblies were 1.38 Gb for Haplotype 1 and 1.40 Gb for Haplotype 2 ( Table 4 , Fig. S3b). The assembled genome size was lower compared to that estimated by GenomeScope 2.0, possibly because of higher repetitive elements in the genome. The 752 primary contigs obtained from the B. tulda HiFiasm assembly were further mapped and compared to other sequenced bamboo genomes. Finally, 43 individual contigs were obtained from the assembled draft genome ( Fig. 4a ). Among them, 39 contigs could be nomenclatured and divided into three distinct (A, B, and C) subgenomes based on their mapping and homology to chromosome-level assemblies of other bamboo genome. This indicated that B. tulda is possibly a hexaploid plant, where each chromosome is represented by three subgenomes. There are some duplicated contigs, which corresponded to a single chromosome in other bamboo genomes, but in this case, they could not be assembled together, and hence assigned as two different contigs. The remaining four contigs could not be assigned to regions of any specific chromosomes. The final draft genome assembly resulted in total genome size of 1.37 Gb, contig N50 value 35.69 Mb, mean contig length 31.77 Mb, and GC content was 44.61% ( Table 5 ). Download figure Open in new tab Fig. 4. Genomic features for the 43 contigs of the B. tulda genome. (a) Circos plot representing the different genome features of the B. tulda genome assembly. The outer to inner tracks represent different contigs, GC content, transposable element density, and collinear blocks. Green, orange, and blue blocks represent contigs of subgenome A, subgenome B, and subgenome C, respectively. Purple blocks represent unidentified contigs. The plot was visualized using Accusyn software [ 50 ]. (b) Dot plot representing the subgenome syntenic relationship of B. tulda based on the annotated coding genes. Annotation of repetitive elements To annotate repetitive elements in B. tulda genome, a combination of de novo and homology-based prediction was used. For de novo prediction, RepeatModeler [ 25 , 26 ] was used, which in turn uses two ab initio repeat sequence prediction softwares: RECON (v1.0.8) [ 27 ] and RepeatScout (v1.0.6)[ 28 ]. LTR_FINDER [ 29 ] was used specifically for de novo prediction of LTR elements. Homology-based method to identify the transposable elements involved using the RepeatMasker [ 30 , 31 ] software against Repbase database. Approximately, 63.31% of the total B. tulda genome was annotated as repeat elements. The identified transposable elements were further classified into two broad categories: Class I retrotransposons and Class II DNA transposons. They represented 27.15% (370 MB) and 6.18% (84 MB) of repetitive sequences, respectively, while 29.97% repeat elements remained unclassified ( Table 7 , Fig. 4a ). Gene prediction and functional annotation For predicting protein-coding genes, a combination of three different methods were applied together: de novo , homology-based, and transcriptome-assisted. The de novo prediction was done using RNA-Seq and protein evidence data, allowing BRAKER3 [ 32 ] to perform automated genome annotation. GeMOMa v1.7 [ 33 ] was used for homology-based prediction, by mapping the previously sequenced bamboo gene models [ 5 ]. The transcriptome guided genome annotation was conducted using Trinity (v2.1.1) [ 34 , 35 ], HISAT2 (v2.2.1) [ 36 ], and StringTie (v1.3.3) [ 37 , 38 ]). The transcriptome data was mapped to the genome using HISAT2 to generate BAM files, which were then used as input for genome-guided transcriptome assembly in StringTie. Also, unigene assemblies were generated by de novo assembly of transcriptome data using Trinity and rnaSPADES [ 39 ]. The genes predicted from all three above methods were consolidated together using Evidence Modeler (EVM) v2.1.0 [ 40 ] and PASA [ 41 , 42 ] to finally obtain non-redundant protein-coding gene structures. In total, 56,890 protein-coding genes constituting 19.44% of the B. tulda genome were identified ( Table 6 ). The predicted genes of B. tulda were mapped back to the B. tulda genome using GeMoMa [ 33 ], with the “synteny checker” module enabled. The dot plot was created with the “synplot.r” script (k=3) included in GeMoMa on the reference gene table created by the GeMoMa run ( Fig. 4b ). For functional annotation, homology-based annotation was done using SwissProt and TrEMBL databases. The predicted protein-coding sequences were aligned against these two databases through BLASTP [ 43 ] analysis with e-value ≤1e −5 . The functional domains of the proteins were identified using InterProScan [ 44 ]. Annotation of rRNA and tRNA Among the non-coding RNA genes, rRNAs and tRNAs were predicted using the B. tulda genome. The rRNA genes were predicted using Barrnap v0.9 [ 45 ], while the tRNA genes were predicted using tRNAscan-SE [ 46 , 47 ]. From this prediction, 1355 rRNA and 1019 tRNA genes were identified successfully ( Table 8 , Fig. 5 ). Download figure Open in new tab Fig. 5. RIdeogram plot to visualize genome-wide gene density, tRNA position, and rRNA position across the 43 assembled contigs of B. tulda . The RIdeogram software was used for visualizing the genome features [ 51 ]. Phylogenetic analysis To determine the evolutionary relationship of B. tulda with other plants, twelve bamboo species ( Ampelocalamus luodianensis , Bonia amplexicaulis , Dendrocalamus latiflorus , D. sinicus , Guadua angustifolia , Hsuehochloa calcarean , Melocanna baccifera , Olyra latifolia , Otatea glauca, P. edulis, Raddia guianensis , and Rhipidocladum racemiflorum ) along with Z. mays and Musa acuminata (banana), were used to construct a species tree. OrthoFinder [ 48 ] was applied to the protein sequences of B. tulda , the twelve sequenced bamboos, and the two outgroups, Z. mays and M. acuminata . From the OrthoFinder analysis 202 orthogroups were obtained which were used for constructing the species tree ( Fig. 6 ). Download figure Open in new tab Fig 6. Species tree of B. tulda and other sequenced bamboos, with Z. mays and M. acuminata used as outgroups. Data Records The raw PacBio genome sequencing data and raw Illumina RNA-Seq data were deposited in the EMBL-EBI European Nucleotide Archive (ENA) database under project accession number PRJEB91656, with sample accession numbers ERS25169745 for PacBio sequencing data and ERS25169746 for Illumina sequencing data. The genome assembly has also been submitted in the EMBL-EBI ENA database and will soon be made publicly available. Technical validation Quality assessment of high molecular weight genomic DNA, libraries and sequence data HMW genomic DNA was extracted to subject it for sequencing in the PacBio platform. The quantity and quality of the extracted DNA was examined by qubit fluorometer, and agarose gel (1%) electrophoresis (Fig. S1a). After ensuring required quality, three SMRTbell libraries were constructed from the genomic DNA, with relative fluorescence units (RFU) peaks obtained at 12,567 bp (DNA1), 11,447 bp (DNA2), and 11,307 bp (DNA3) (Figs. S1b, c, d). Each of these 3 libraries were sequenced in the PacBio Sequel IIe platform, while libraries DNA1 and DNA2 were sequenced in the PacBio Revio platform. The mean read lengths obtained in the Sequel IIe platform were 12,472 bp (DNA1), 12,524 bp (DNA2), and 12,864 bp (DNA3), while in the Revio platform were 10,033 bp (DNA1) and 13,155 bp (DNA2) (Fig. S2). Assessment of genome assembly and annotation completeness The completeness of the assembled genome was examined in BUSCO v5.8.0 [ 49 ] against the Embryophyte (n = 425) and Viridiplantae (n = 1614) datasets. BUSCO analysis demonstrated 99.0% complete, 4% fragmented, and 0% missing BUSCO groups against Embryophyte, and 98.0% complete, 3% fragmented, and 2% missing BUSCO groups against Viridiplantae ( Fig. 3 ). Code availability All softwares and pipeline utilized in this study for data analyses were implemented in full compliance with the manuals and protocols described by the respective published bioinformatics tools. Details on the software versions and parameters are outlined in the Methods section. In cases where specific parameters are not specified, default settings were applied. No custom programming or coding was employed. Author contributions MD and AB conceived the study, acquired funding and designed experimental plan. SK, SD, MB, and SB were involved in collection of plant samples, performing FACS experiments for genome size estimation, and isolating high molecular weight genomic DNA for high-throughput sequencing. SK and OR conducted all in silico experiments on genome data analysis. SK wrote the manuscript, and MD, AB and OR edited it. All authors read and approved the final manuscript. Competing interests The authors declare no competing interests. Acknowledgement The research results reported in this paper are funded by the Alexander von Humboldt Foundation, Germany through ‘Research Group Linkage Program’. SK and SB acknowledge JRF fellowships from UGC, India, with NTA Ref. no.: 231610225668 and 211610047259 respectively. SD acknowledges CSIR Project (Grant no.: 38(1493)/19/EMR-II). We thank Prof. J. Dolezel for providing us the seeds of internal reference plants for genome size estimation. We thank Subhayan Paul, Institute of Health Sciences, Presidency University and Rajesh Saha, Bio-Rad Laboratories for providing technical support in operating the flow cytometer machine. Funder Information Declared Alexander von Humboldt Foundation , Ref 3.4 - 1131176 - IND - IP Footnotes https://www.ebi.ac.uk/ena/browser/view/PRJEB91656 References 1. ↵ Basak , M. , Dutta , S. , Biswas , S. , Chakraborty , S. , Sarkar , A. , Rahaman , T. , Dey , S. , Biswas , P. , & Das , M . ( 2021 ). Genomic insights into growth and development of bamboos: what have we learnt and what more to discover? . Trees , 35 , 1771 – 1791 . OpenUrl 2. ↵ Zheng , Y. , Yang , D. , Rong , J. , Chen , L. , Zhu , Q. , He , T. , & Gu , L . ( 2022 ). Allele□aware chromosome[scale assembly of the allopolyploid genome of hexaploid Ma Bamboo ( Dendrocalamus latiflorus Munro) . Journal of Integrative Plant Biology , 64 ( 3 ), 649 – 670 . OpenUrl PubMed 3. ↵ Kellogg , E. A. , & Kellogg , E. A . ( 2015 ). Description of the Family, Vegetative Morphology and Anatomy: Poaceae (R. Br.) Barnh.(1895). Gramineae Juss.(1789) . Flowering Plants. Monocots: Poaceae , 3 – 23 . 4. ↵ Zhao , H. , Wang , J. , Meng , Y. , Li , Z. , Fei , B. , Das , M. , & Jiang , Z . ( 2022 ). Bamboo and rattan: Nature-based solutions for sustainable development . The Innovation , 3 ( 6 ). 5. ↵ Ma , P. F. , Liu , Y. L. , Guo , C. , Jin , G. , Guo , Z. H. , Mao , L. , & Li , D. Z . ( 2024 ). Genome assemblies of 11 bamboo species highlight diversification induced by dynamic subgenome dominance . Nature Genetics , 56 ( 4 ), 710 – 720 . OpenUrl CrossRef PubMed 6. ↵ Li extra spacing as dots, instead put names , W., Shi , C. , Li , K. , Zhang , Q. J. , Tong , Y. , Zhang , Y. , … & Gao , L. Z. ( 2021 ). Draft genome of the herbaceous bamboo Raddia distichophylla . G3 , 11 ( 2 ), jkaa049 . OpenUrl 7. ↵ Janzen , D. H . ( 1976 ). Why bamboos wait so long to flower . Annual Review of Ecology and systematics , 347 – 391 . 8. Chakraborty , S. , Biswas , P. , Dutta , S. , Basak , M. , Guha , S. , Chatterjee , U. , & Das , M . ( 2021 ). Studies on reproductive development and breeding habit of the commercially important bamboo bambusa tulda roxb . Plants , 10 ( 11 ), 2375 . OpenUrl PubMed 9. ↵ Biswas , S. , Rahaman , T. , Gupta , P. , Mitra , R. , Dutta , S. , Kharlyngdoh , E. , Guha , S. , Ganguly , J. , Pal , A. , & Das , M . ( 2022 ). Cellulose and lignin profiling in seven, economically important bamboo species of India by anatomical, biochemical, FTIR spectroscopy and thermogravimetric analysis . Biomass and Bioenergy , 158 , 106362 . OpenUrl 10. ↵ Peng , Z. , Lu , Y. , Li , L. , Zhao , Q. , Feng , Q. I. , Gao , Z. , … & Jiang , Z. ( 2013 ). The draft genome of the fast-growing non-timber forest species moso bamboo (Phyllostachys heterocycla) . Nature genetics , 45 ( 4 ), 456 – 461 . OpenUrl CrossRef PubMed 11. ↵ Espinosa , E. , Bautista , R. , Larrosa , R. , & Plata , O . ( 2024 ). Advancements in long-read genome sequencing technologies and algorithms . Genomics , 110842 . 12. ↵ Kumar , P. P. , Turner , I. M. , Nagaraja Rao , A. , & Arumuganathan , K . ( 2011 ). Estimation of nuclear DNA content of various bamboo and rattan species . Plant Biotechnology Reports , 5 , 317 – 322 . OpenUrl 13. ↵ Ram , H. M. , & Gopal , B. H . ( 1981 ). Some observations on the flowering of bamboos in Mizoram . Current Science , 708 – 710 . 14. ↵ Bhattacharya , S. , Das , M. , Bar , R. , & Pal , A . ( 2006 ). Morphological and molecular characterization of Bambusa tulda with a note on flowering . Annals of Botany , 98 ( 3 ), 529 – 535 . OpenUrl CrossRef PubMed 15. ↵ Das , M. , Bhattacharya , S. , & Pal , A . ( 2005 ). Generation and characterization of SCARs by cloning and sequencing of RAPD products: a strategy for species-specific marker development in bamboo . Annals of botany , 95 ( 5 ), 835 – 841 . OpenUrl CrossRef PubMed 16. ↵ Das , M. , & Pal , A . ( 2005 ). Clonal propagation and production of genetically uniform regenerants from axillary meristems of adult bamboo . Journal of Plant biochemistry and biotechnology , 14 , 185 – 188 . OpenUrl 17. ↵ Biswas , P. , Chakraborty , S. , Dutta , S. , Pal , A. , & Das , M . ( 2016 ). Bamboo flowering from the perspective of comparative genomics and transcriptomics . Frontiers in plant science , 7 , 1900 . OpenUrl PubMed 18. Dutta , S. , Biswas , P. , Chakraborty , S. , Mitra , D. , Pal , A. , & Das , M . ( 2018 ). Identification, characterization and gene expression analyses of important flowering genes related to photoperiodic pathway in bamboo . BMC genomics , 19 , 1 – 19 . OpenUrl CrossRef PubMed 19. Dutta , S. , Deb , A. , Biswas , P. , Chakraborty , S. , Guha , S. , Mitra , D. , & Das , M . ( 2021 ). Identification and functional characterization of two bamboo FD gene homologs having contrasting effects on shoot growth and flowering . Scientific Reports , 11 ( 1 ), 7849 . OpenUrl PubMed 20. ↵ Basak , M. , Chakraborty , S. , Kundu , S. , Dey , S. , & Das , M . ( 2025 ). Identification, expression analyses of APETALA1 gene homologs in Bambusa tulda and heterologous validation of BtMADS14 in Arabidopsis thaliana . Physiology and Molecular Biology of Plants , 1-16. 21. ↵ Doležel , J. , Greilhuber , J. , & Suda , J . ( 2007 ). Estimation of nuclear DNA content in plants using flow cytometry . Nature protocols , 2 ( 9 ), 2233 – 2244 . OpenUrl PubMed 22. ↵ Marçais , G. , & Kingsford , C . ( 2011 ). A fast, lock-free approach for efficient parallel counting of occurrences of k-mers . Bioinformatics , 27 ( 6 ), 764 – 770 . OpenUrl CrossRef PubMed Web of Science 23. ↵ Ranallo-Benavidez , T. R. , Jaron , K. S. , & Schatz , M. C . ( 2020 ). GenomeScope 2.0 and Smudgeplot for reference-free profiling of polyploid genomes . Nature communications , 11 ( 1 ), 1432 . OpenUrl PubMed 24. ↵ Cheng , H. , Concepcion , G. T. , Feng , X. , Zhang , H. , & Li , H . ( 2021 ). Haplotype-resolved de novo assembly using phased assembly graphs with hifiasm . Nature methods , 18 ( 2 ), 170 – 175 . OpenUrl CrossRef PubMed 25. ↵ Smit , A. F. , Hubley , R. , & Green , P . ( 2008 ). RepeatModeler Open-1.0 . 26. ↵ Flynn , J. M. , Hubley , R. , Goubert , C. , Rosen , J. , Clark , A. G. , Feschotte , C. , & Smit , A. F . ( 2020 ). RepeatModeler2 for automated genomic discovery of transposable element families . Proceedings of the National Academy of Sciences , 117 ( 17 ), 9451 – 9457 . OpenUrl Abstract / FREE Full Text 27. ↵ Bao , Z. , & Eddy , S. R . ( 2002 ). Automated de novo identification of repeat sequence families in sequenced genomes . Genome research , 12 ( 8 ), 1269 – 1276 . OpenUrl Abstract / FREE Full Text 28. ↵ Price , A. L. , Jones , N. C. , & Pevzner , P. A . ( 2005 ). De novo identification of repeat families in large genomes . Bioinformatics , 21 ( suppl_1 ), i351 – i358 . OpenUrl CrossRef PubMed Web of Science 29. ↵ Xu , Z. , & Wang , H . ( 2007 ). LTR_FINDER: an efficient tool for the prediction of full-length LTR retrotransposons . Nucleic acids research , 35 ( suppl_2 ), W265 – W268 . OpenUrl CrossRef PubMed Web of Science 30. ↵ Tarailo-Graovac , M. , & Chen , N . ( 2009 ). Using RepeatMasker to identify repetitive elements in genomic sequences. Current protocols in bioinformatics , Chapter 4 , 4.10. 1–4.10. 14. 31. ↵ Smit , A. F. A. , Hubley , R. , & Green , P. (2013– 2015 ). RepeatMasker Open-4.0 . 32. ↵ Gabriel , L. , Brůna , T. , Hoff , K. J. , Ebel , M. , Lomsadze , A. , Borodovsky , M. , & Stanke , M . ( 2024 ). BRAKER3: Fully automated genome annotation using RNA-seq and protein evidence with GeneMark-ETP, AUGUSTUS, and TSEBRA . Genome Research , 34 ( 5 ), 769 – 777 . OpenUrl Abstract / FREE Full Text 33. ↵ Keilwagen , J. , Hartung , F. , & Grau , J . ( 2019 ). GeMoMa: homology-based gene prediction utilizing intron position conservation and RNA-seq data . Gene prediction: Methods and protocols , 161 – 177 . 34. ↵ Grabherr , M. G. , Haas , B. J. , Yassour , M. , Levin , J. Z. , Thompson , D. A. , Amit , I. , Adiconis , X. , Fan , L. , Raychaowdhury , R. , Zeng , Q. , & Regev , A. ( 2011 ). Trinity: reconstructing a full-length transcriptome without a genome from RNA-Seq data . Nature biotechnology , 29 ( 7 ), 644 . OpenUrl CrossRef PubMed 35. ↵ Haas , B. J. , Papanicolaou , A. , Yassour , M. , Grabherr , M. , Blood , P. D. , Bowden , J. , Couger , M. B. , Eccles , D. , Li , B. O. , Lieber , M. , & Regev , A . ( 2013 ). De novo transcript sequence reconstruction from RNA-seq using the Trinity platform for reference generation and analysis . Nature protocols , 8 ( 8 ), 1494 – 1512 . OpenUrl PubMed 36. ↵ Kim , D. , Langmead , B. , & Salzberg , S. L . ( 2015 ). HISAT: a fast spliced aligner with low memory requirements . Nature methods , 12 ( 4 ), 357 – 360 . OpenUrl CrossRef PubMed 37. ↵ Pertea , M. , Pertea , G. M. , Antonescu , C. M. , Chang , T. C. , Mendell , J. T. , & Salzberg , S. L . ( 2015 ). StringTie enables improved reconstruction of a transcriptome from RNA-seq reads . Nature biotechnology , 33 ( 3 ), 290 – 295 . OpenUrl CrossRef PubMed 38. ↵ Kovaka , S. , Zimin , A. V. , Pertea , G. M. , Razaghi , R. , Salzberg , S. L. , & Pertea , M . ( 2019 ). Transcriptome assembly from long-read RNA-seq alignments with StringTie2 . Genome biology , 20 , 1 – 13 . OpenUrl CrossRef PubMed 39. ↵ Bushmanova , E. , Antipov , D. , Lapidus , A. , & Prjibelski , A. D . ( 2019 ). rnaSPAdes: a de novo transcriptome assembler and its application to RNA-Seq data . GigaScience , 8 ( 9 ), giz100 . OpenUrl CrossRef PubMed 40. ↵ Haas , B. J. , Salzberg , S. L. , Zhu , W. , Pertea , M. , Allen , J. E. , Orvis , J. , White , O. , Buell , C. R. , & Wortman , J. R . ( 2008 ). Automated eukaryotic gene structure annotation using EVidenceModeler and the Program to Assemble Spliced Alignments . Genome biology , 9 , 1 – 22 . OpenUrl CrossRef PubMed 41. ↵ Haas , B. J. , Delcher , A. L. , Mount , S. M. , Wortman , J. R. , Smith Jr , R. K. , Hannick , L. I. , Maiti , R. , Ronning , C. M. , Rusch , D. B. , Town , C. D. , & White , O . ( 2003 ). Improving the Arabidopsis genome annotation using maximal transcript alignment assemblies . Nucleic acids research , 31 ( 19 ), 5654 – 5666 . OpenUrl CrossRef PubMed Web of Science 42. ↵ Campbell , M. A. , Haas , B. J. , Hamilton , J. P. , Mount , S. M. , & Buell , C. R . ( 2006 ). Comprehensive analysis of alternative splicing in rice and comparative analyses with Arabidopsis . BMC genomics , 7 , 1 – 17 . OpenUrl CrossRef PubMed Web of Science 43. ↵ Altschul , S. F. , Madden , T. L. , Schäffer , A. A. , Zhang , J. , Zhang , Z. , Miller , W. , & Lipman , D. J . ( 1997 ). Gapped BLAST and PSI-BLAST: a new generation of protein database search programs . Nucleic acids research , 25 ( 17 ), 3389 – 3402 . OpenUrl CrossRef PubMed Web of Science 44. ↵ Jones , P. , Binns , D. , Chang , H. Y. , Fraser , M. , Li , W. , McAnulla , C. , McWilliam , H. , Maslen , J. , Litchell , a. , Nuka , G. , & Hunter , S. ( 2014 ). InterProScan 5: genome-scale protein function classification . Bioinformatics , 30 ( 9 ), 1236 – 1240 . OpenUrl CrossRef PubMed Web of Science 45. ↵ Seemann , T . ( 2013 ). barrnap 0.9: rapid ribosomal RNA prediction . Google Scholar , 792 . 46. ↵ Chan , P. P. , & Lowe , T. M. ( 2019 ). tRNAscan-SE: searching for tRNA genes in genomic sequences (pp. 1 – 14 ). Springer New York . 47. ↵ Chan , P. P. , Lin , B. Y. , Mak , A. J. , & Lowe , T. M . ( 2021 ). tRNAscan-SE 2.0: improved detection and functional classification of transfer RNA genes . Nucleic acids research , 49 ( 16 ), 9077 – 9096 . OpenUrl CrossRef PubMed 48. ↵ Emms , D. M. , & Kelly , S . ( 2019 ). OrthoFinder: phylogenetic orthology inference for comparative genomics . Genome biology , 20 , 1 – 14 . OpenUrl CrossRef PubMed 49. ↵ Simão , F. A. , Waterhouse , R. M. , Ioannidis , P. , Kriventseva , E. V. , & Zdobnov , E. M . ( 2015 ). BUSCO: assessing genome assembly and annotation completeness with single-copy orthologs . Bioinformatics , 31 ( 19 ), 3210 – 3212 . OpenUrl CrossRef PubMed 50. ↵ Siri , J. N. , Neufeld , E. , Parkin , I. , & Sharpe , A. ( 2020 , May). Using Simulated Annealing to Declutter Genome Visualizations . In FLAIRS (pp. 201 – 204 ). 51. ↵ Hao , Z. , Lv , D. , Ge , Y. , Shi , J. , Weijers , D. , Yu , G. , & Chen , J . ( 2020 ). RIdeogram: drawing SVG graphics to visualize and map genome-wide data on the idiograms . PeerJ Computer Science , 6 , e251 . OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted July 22, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Genome sequencing, de novo assembly and annotation of the commercially important bamboo, Bambusa tulda Roxb Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Genome sequencing, de novo assembly and annotation of the commercially important bamboo, Bambusa tulda Roxb Sutrisha Kundu , Oliver Rupp , Sonali Dey , Mridushree Basak , Sudeshna Bera , Annette Becker , Malay Das bioRxiv 2025.07.19.665512; doi: https://doi.org/10.1101/2025.07.19.665512 Share This Article: Copy Citation Tools Genome sequencing, de novo assembly and annotation of the commercially important bamboo, Bambusa tulda Roxb Sutrisha Kundu , Oliver Rupp , Sonali Dey , Mridushree Basak , Sudeshna Bera , Annette Becker , Malay Das bioRxiv 2025.07.19.665512; doi: https://doi.org/10.1101/2025.07.19.665512 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genomics Subject Areas All Articles Animal Behavior and Cognition (7618) Biochemistry (17636) Bioengineering (13860) Bioinformatics (41847) Biophysics (21401) Cancer Biology (18536) Cell Biology (25424) Clinical Trials (138) Developmental Biology (13353) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24287) Genetics (15583) Genomics (22463) Immunology (17701) Microbiology (40300) Molecular Biology (17141) Neuroscience (88434) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4813) Physiology (7633) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4285) Systems Biology (9808) Zoology (2268)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.