Chromosome-scale genome assembly and annotation of the two-spotted cricket Gryllus bimaculatus (Orthoptera: Gryllidae)

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

The two-spotted cricket, Gryllus bimaculatus , is a key hemimetabolous model organism for developmental biology, neuroscience, and regeneration. The existing reference genome is, however, highly fragmented into 47,877 scaffolds, hampering chromosome-scale analyses for these fields. Here, we report a high-quality, chromosome-scale genome assembly for the white-eyed mutant strain of this cricket, generated using a combination of Nanopore and PacBio HiFi long reads, integrated with Hi-C data. The final 1.62 Gbp assembly achieves a scaffold N50 of 107.4 Mbp, a significant improvement in contiguity over the previous 6.3 Mbp N50. We anchored 94.45% of the assembly into 15 pseudomolecules, consistent with the known karyotype (n = 15). The genome completeness (BUSCO v6.0.0 insecta_odb12) reached 98.1%. We also updated the annotation, identifying 14,964 protein-coding genes. This gene set shows markedly improved completeness (BUSCO v6.0.0 insecta_odb12: 95.7%) compared with the previous annotation (81.2%) and successfully recovers all nine essential neuropeptide genes previously reported as missing from the draft assembly. This chromosome-scale genomic resource provides an essential foundation for comparative and functional genomics in G. bimaculatus .
Full text 43,136 characters · extracted from preprint-html · click to expand
Chromosome-scale genome assembly and annotation of the two-spotted cricket Gryllus bimaculatus (Orthoptera: Gryllidae) | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Chromosome-scale genome assembly and annotation of the two-spotted cricket Gryllus bimaculatus (Orthoptera: Gryllidae) View ORCID Profile Kosuke Kataoka , Ryuto Sanno , View ORCID Profile Tomasz Gaczorek , View ORCID Profile Upendra Raj Bhattarai , Yuki Ito , Shintaro Inoue , View ORCID Profile Kei Yura , Toru Asahi , View ORCID Profile Guillem Ylla , View ORCID Profile Taro Mito , View ORCID Profile Cassandra G. Extavour doi: https://doi.org/10.1101/2025.10.31.685973 Kosuke Kataoka 1 Institute of Engineering, Tokyo University of Agriculture and Technology , Tokyo, Japan 2 Comprehensive Research Organization, Waseda University , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kosuke Kataoka For correspondence: kataokak{at}go.tuat.ac.jp guillem.ylla{at}uj.edu.pl mito.taro{at}tokushima-u.ac.jp extavour{at}oeb.harvard.edu Ryuto Sanno 3 Graduate School of Advanced Science and Engineering, Waseda University , Tokyo, Japan 4 Faculty of Biochemistry, Biophysics and Biotechnology, Jagiellonian University , Kraków, Poland Find this author on Google Scholar Find this author on PubMed Search for this author on this site Tomasz Gaczorek 4 Faculty of Biochemistry, Biophysics and Biotechnology, Jagiellonian University , Kraków, Poland Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tomasz Gaczorek Upendra Raj Bhattarai 6 Harvard T.H. Chan School of Public Health, Harvard University , Boston, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Upendra Raj Bhattarai Yuki Ito 3 Graduate School of Advanced Science and Engineering, Waseda University , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Shintaro Inoue 7 Bio-Innovation Research Center, Tokushima University , Tokushima, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kei Yura 2 Comprehensive Research Organization, Waseda University , Tokyo, Japan 3 Graduate School of Advanced Science and Engineering, Waseda University , Tokyo, Japan 8 Graduate School of Humanities and Sciences, Ochanomizu University , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kei Yura Toru Asahi 2 Comprehensive Research Organization, Waseda University , Tokyo, Japan 3 Graduate School of Advanced Science and Engineering, Waseda University , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Guillem Ylla 4 Faculty of Biochemistry, Biophysics and Biotechnology, Jagiellonian University , Kraków, Poland Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Guillem Ylla For correspondence: kataokak{at}go.tuat.ac.jp guillem.ylla{at}uj.edu.pl mito.taro{at}tokushima-u.ac.jp extavour{at}oeb.harvard.edu Taro Mito 7 Bio-Innovation Research Center, Tokushima University , Tokushima, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Taro Mito For correspondence: kataokak{at}go.tuat.ac.jp guillem.ylla{at}uj.edu.pl mito.taro{at}tokushima-u.ac.jp extavour{at}oeb.harvard.edu Cassandra G. Extavour 5 Department of Organismic and Evolutionary Biology, Harvard University , Cambridge, MA, USA 9 Department of Molecular and Cellular Biology, Harvard University , Cambridge, MA, USA 10 Howard Hughes Medical Institute , Chevy Chase, MD, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Cassandra G. Extavour For correspondence: kataokak{at}go.tuat.ac.jp guillem.ylla{at}uj.edu.pl mito.taro{at}tokushima-u.ac.jp extavour{at}oeb.harvard.edu Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract The two-spotted cricket, Gryllus bimaculatus , is a key hemimetabolous model organism for developmental biology, neuroscience, and regeneration. The existing reference genome is, however, highly fragmented into 47,877 scaffolds, hampering chromosome-scale analyses for these fields. Here, we report a high-quality, chromosome-scale genome assembly for the white-eyed mutant strain of this cricket, generated using a combination of Nanopore and PacBio HiFi long reads, integrated with Hi-C data. The final 1.62 Gbp assembly achieves a scaffold N50 of 107.4 Mbp, a significant improvement in contiguity over the previous 6.3 Mbp N50. We anchored 94.45% of the assembly into 15 pseudomolecules, consistent with the known karyotype (n = 15). The genome completeness (BUSCO v6.0.0 insecta_odb12) reached 98.1%. We also updated the annotation, identifying 14,964 protein-coding genes. This gene set shows markedly improved completeness (BUSCO v6.0.0 insecta_odb12: 95.7%) compared with the previous annotation (81.2%) and successfully recovers all nine essential neuropeptide genes previously reported as missing from the draft assembly. This chromosome-scale genomic resource provides an essential foundation for comparative and functional genomics in G. bimaculatus . Introduction The two-spotted cricket, Gryllus bimaculatus , is a model organism for hemimetabolous insects ( Horch et al., 2017 ) ( Fig. 1 ). Unlike holometabolous insects such as the fruit fly ( Drosophila melanogaster ) or the red flour beetle ( Tribolium castaneum ), hemimetabolous insects undergo direct development, where nymphs hatch and grow through successive molts to become adults without larval or pupal stages. This ancestral developmental mode is crucially important for understanding insect evolution. Due to its ease of rearing, short generation time, and the applicability of efficient RNA interference (RNAi) ( Miyawaki et al., 2004 ) and genome-editing techniques such as CRISPR/Cas9 ( Matsuoka et al., 2025 ), G. bimaculatus has been established as a powerful experimental model in a wide range of fields, including evolutionary developmental biology ( Donoughe & Extavour, 2016 ), regeneration biology ( Nakamura et al., 2008 ), neuroscience ( Matsumoto et al., 2018 ), and ethology ( Abe et al., 2021 ; Kuriwada, 2022 ). It is also gaining significant global attention as a novel, sustainable food source due to its high protein content and efficient rearing ( Kataoka et al., 2020 , 2022 ; Mito et al., 2022 ). Download figure Open in new tab Figure 1. Adult male (left) and female (right) of the white-eyed mutant strain G. bimaculatus . In the first report of a genomic resource for G. bimaculatus , the white-eyed mutant strain, which is standardly used in many functional studies, was sequenced ( Ylla et al., 2021 ). The assembly (GenBank accession: GCA_017312745.1) was a useful resource with a genome size of approximately 1.66 Gb and a scaffold N50 of 6.3 Mb, contributing to analyses of gene family evolution and DNA methylation. However, this previous assembly was fragmented into 47,877 scaffolds and was not assembled to chromosome scale. While chromosome-scale genomes have recently been reported for other related cricket species, such as Gryllus assimilis ( Ito et al., 2025 ) and Acheta domesticus ( Dossey et al., 2023 ), such a resource has remained unavailable for G. bimaculatus , which serves as the primary model for functional, developmental, and neurobiological studies. The lack of a chromosome-scale assembly for this specific species has been a significant limitation for large-scale comparative genomic analyses based on synteny (conservation of gene order), understanding the structural arrangement of transposons and repetitive sequences on chromosomes, and for quantitative trait locus mapping and the accurate identification of genome-editing off-target sites. In this study, to fill the gap, we constructed a high-quality, chromosome-scale genome assembly using the same, white-eyed mutant strain, used in the previous study. By combining Nanopore and PacBio HiFi long-read sequencing methods, and Hi-C chromatin conformation capture technology, we anchored 94.45% of the entire genome into 15 pseudomolecules, corresponding to the G. bimaculatus karyotype (n = 14 autosomes + X) ( Yoshimura et al., 2006 ). This assembly achieves a contig N50 of 4.58 Mb and a scaffold N50 of 107.39 Mb, representing a significant improvement in contiguity. Furthermore, we report an updated gene annotation comprising 14,964 protein-coding genes. This chromosome-scale genome resource will strengthen the foundation for all genomic research using G. bimaculatus and accelerate new insights into the biology of hemimetabolous insects. Materials and Methods Animals A white-eyed mutant strain of G. bimaculatus ( Mito & Noji, 2008 ) was housed in plastic cases at 30 °C ± 1 °C and 30%–40% relative humidity under a 10 h light and 14 h dark photoperiod. They were nourished with an artificial fish food (4971618–011312, Kyorin, Japan). Library preparation and sequencing A single alive G. bimaculatus individual ( Fig. 1 ) was used for genomic DNA extraction. Total genomic DNA was extracted from the head and hind legs of a male G. bimaculatus using NucleoBond® HMW DNA (Macherey-Nagel, Germany) according to the manufacturer’s instructions. The resulting genomic DNA was size-selected using a Short Read Eliminator Kit (PacBio, CA, USA). DNA purity and concentrations were measured by spectrometry using NanoPhotometer NP80-TOUCH (Implen, Germany) and fluorometry using Qubit 4 (Thermo Fisher Scientific, MA, USA). For long-read sequencing, Oxford Nanopore Technologies (ONT) libraries were constructed using the Ligation Sequencing Kit V14 and sequenced on the PromethION 2 Solo platform (Oxford Nanopore Technologies, UK) with a Flow Cell R10.4.1. Base-calling was performed using Dorado v0.3.0 (model: [email protected] ). The resulting raw reads were adapter-trimmed using Porechop_ABI v0.5.0 ( Bonenfant et al., 2023 ). Additionally, a SMRTbell library was prepared and sequenced on a PacBio Sequel IIe system. For chromosome-scale scaffolding, two separate Hi-C libraries were prepared. The first library was prepared from the hind legs of a single male G. bimaculatus using the Dovetail™ Omni-C™ Kit (Dovetail Genomics, CA, USA) following the manufacturer’s instructions. The second Hi-C library was generated from the thorax and legs of an adult male using the Proximo Hi-C (Animal) kit (KT2045) (Phase Genomics, Seattle, US), following the manufacturer’s instructions. Both libraries were sequenced on the Illumina NovaSeq 6000 platform, and the sequencing data were combined for downstream scaffolding analysis. Genome de novo assembly The initial draft genome was assembled by combining the filtered ONT long reads and the PacBio HiFi reads using Flye v2.9.5 ( Kolmogorov et al., 2019 ). To correct errors in the resulting contigs, we retrieved publicly available Illumina short-read data (DDBJ Sequence Read Archive [DRA] accessions DRR272308–DRR272313), which were originally sequenced on an Illumina HiSeq 2000. These reads were downsampled to an approximate 93.9x coverage and used for polishing. The assembly underwent two rounds of error correction using POLCA v4.1.0 ( Zimin & Salzberg, 2020 ) with default settings. Potential contamination in the assembly was removed using BlobToolKit v1.1.1 ( Challis et al., 2020 ), which analyzes unexpected coverage, GC content, or similarity to bacterial and other contaminant sequences. Sequence coverage was determined by mapping Illumina reads with bwa v0.7.17-r1188. Similarity analysis was performed using BLASTn v2.13.0+ against NCBI NT database v5 (options: - task megablast culling_limit 10 -evalue 1e-25 -outfmt ‘6 qseqid staxids bitscore std sscinames sskingdoms stitle’). Mitochondrial genomes were also identified through gene prediction using the MITOS2 webserver and subsequently removed. The resulting contigs were then corrected for misjoins, ordered, oriented, and anchored into a chromosome-scale assembly using Omni-C™ data with Juicer v1.9.9 ( Durand et al., 2016 ) and 3D-DNA v180419 ( Dudchenko et al., 2017 ). Candidate assembly was reviewed with Juicebox Assembly Tools v1.9.9 ( Durand et al., 2016 ) for quality control and interactive corrections. The contact map was visualized using Juicebox. The completeness of the final genome assembly was assessed using BUSCO v6.0.0 ( Tegenfeldt et al., 2025 ) against the insecta_odb12 lineage datasets. Prediction of repeat regions In de novo repeat prediction, RepeatModeler v2.0.6 ( Flynn et al., 2020 ) was first used for de novo repeat identification. Because standard libraries are insufficient for effective masking in Gryllus genomes ( Szrajer et al., 2024 ), this library was supplemented with the custom repeat library previously generated ( Ylla et al., 2021 ). The combined repeat library was then used to identify and softmask repetitive elements in the G. bimaculatus genome using RepeatMasker v4.2.1 ( Smit et al., 2015 ). Structural gene annotation Structural annotation for protein-coding genes was performed on the softmasked genome using ab initio prediction and RNA-seq-based prediction. Both methods used publicly available RNA-seq data (Sequence Read Archive [SRA] accessions: SRR10619411, SRR10619415, SRR10619417, SRR10619418, SRR10619421, SRR10619423, SRR10619425, SRR10619429, SRR10619431, SRR10619432, SRR10619434, SRR10619437, SRR10619439, SRR10619440, SRR14026720– SRR14026726) as input. To remove noisy RNA-seq reads potentially arising from erroneous transcription and splicing, de novo transcriptome assembly was first performed using Trinity v2.15.1 ( Haas et al., 2013 ) to generate contigs. The original RNA-seq reads were then mapped back to these contigs using HISAT2 v2.2.1 ( Kim et al., 2019 ) with default parameters, allowing filtration of reads that did not map correctly in the proper orientation as paired-end reads. After removing these noisy reads, the remaining reads were subsequently used for gene predictions. The ab initio prediction was carried out using BRAKER v3.0.8 ( Brůna et al., 2021 ; Gabriel et al., 2024 ; Hoff et al., 2016 , 2019 ; Stanke et al., 2008 , 2006 ), incorporating protein data from the OrthoDB 12 arthropods dataset (retrieved from https://bioinf.uni-greifswald.de/bioinf/partitioned_odb12/ ) and the mapping data of the filtered RNA-seq reads. This BRAKER prediction served as the foundation for our gene set. To complement this base set, StringTie2 v2.2.1 ( Kovaka et al., 2019 ) was used for RNA-seq-based prediction. These combined predictions (BRAKER and StringTie2) were merged and duplicate genes were discarded using GffCompare v0.12.6 ( Pertea & Pertea, 2020 ) to form a final, comprehensive consensus gene set. Functional gene annotation Gene functional annotation was conducted using eggNOG-mapper online ( http://eggnog-mapper.embl.de/ ) ( Cantalapiedra et al., 2021 ) and BLASTp-based methods. For the BLASTp-based annotation, we used databases including Homo sapiens, Mus musculus, Caenorhabditis elegans, D. melanogaster , and UniProt Swiss-Prot to identify the best hits for annotation (E-value < 1.0 × 10 −10 ). The completeness of the final predicted gene models was assessed using BUSCO v6.0.0 ( Tegenfeldt et al., 2025 ) against the insecta_odb12 lineage dataset. For this analysis, the longest isoform for each gene was first extracted from the annotation file using AGAT v0.9.1 ( Dainat et al., 2022 ). Validation of neuropeptide gene loci To assess the completeness of our assembly regarding functionally important gene families, we specifically investigated the neuropeptide gene loci previously reported ( Mochizuki et al., 2023 ). The neuropeptide cDNA sequences listed in the study were retrieved. These sequences were mapped against our final chromosome-scale genome assembly using Exonerate v2.4.0 ( Slater & Birney, 2005 ) with the est2genome model to accurately determine exon-intron boundaries. All resulting alignments were manually inspected to validate the gene structures. Results and Discussion Genome sequencing To construct the genome assembly, we generated three types of sequencing data ( Table 1 ). First, we obtained 45.00 Gbp of ONT sequencing data; the average and N50 read lengths were 13.13 Kbp and 24.23 Kbp, respectively. Second, we generated 13.44 Gbp of PacBio HiFi data, with an average read length of 13.61 Kbp and an N50 of 14.17 Kbp. Finally, for chromosome-scale scaffolding, 243.22 Gbp of Hi-C raw data was generated from the Illumina platform. View this table: View inline View popup Download powerpoint Table 1. Statistics for the DNA-seq data of the G. bimaculatus genome. Genome de novo assembly statistics The hybrid assembly strategy combining ONT and HiFi reads, followed by two rounds of Illumina-based polishing, yielded a 1.63 Gbp draft genome. This polished assembly comprised 3,789 contigs and achieved a contig N50 of 4.58 Mbp. Following Hi-C-based scaffolding, the final assembly resulted in a genome of 1.62 Gbp in size, visualized by a snail plot ( Fig. 2A ). This assembly consists of 196 scaffolds with a scaffold N50 length of 107 Mbp ( Table 2 ). This represents a substantial improvement in contiguity compared to the previous G. bimaculatus assembly ( Ylla et al., 2021 ), which had a scaffold N50 of 6.3 Mbp and was fragmented into 47,877 scaffolds ( Fig. 2B ). To assess genomic completeness, we performed a BUSCO v6.0.0 analysis using the insecta_odb12 dataset. Our assembly achieved a completeness score of 98.1% (C:98.1%[S:96.0%,D:2.2%],F:0.4%,M:1.5%), demonstrating a higher level of completeness compared to the 96.0% (C:96.0%[S:94.3%,D:1.7%],F:1.4%,M:2.6%) of the previous genome ( Table 3 ). View this table: View inline View popup Download powerpoint Table 2. Summary statistics for the chromosome-scale assembly of G. bimaculatus . View this table: View inline View popup Download powerpoint Table 3. BUSCO completeness assessment of the G. bimaculatus genome assembly. Download figure Open in new tab Figure 2. Assembly statistics and contiguity of the Gryllus bimaculatus (white-eyed strain) chromosome-scale genome. (A) Snail plot visualizing the key statistics of the final assembly. The cumulative assembly length (1.62 Gbp) is plotted in light orange, with the longest scaffold (254.5 Mbp) shown in red. The N50 length (107.4 Mbp) is indicated by the orange arc. The inner purple spiral plots the cumulative number of scaffolds (196 total) on a log scale, with white scale lines drawn at successive orders of magnitude from 10 scaffolds onwards. The circumferential axis indicates the base composition of the assembly (GC: 40.4%, AT: 59.6%, N: 0.1%). The donut chart (top right) displays the BUSCO completeness (insecta_odb12), showing 98.1% total complete (“Comp.”) genes (light green), which includes 96.0% single-copy and 2.2% duplicated (“Dup.”) genes (dark green). An additional 0.4% were fragmented (“Frag.”) genes (pale green). An interactive version is available at https://kataokaklab.github.io/snailplot-assembly-stats/ . (B) Cumulative length plot comparing the contiguity of this assembly (blue line) with the previous draft assembly ( Ylla et al., 2021 ) (orange line). The y-axis shows the percentage of the total genome length covered, and the x-axis (log scale) shows the number of scaffolds. A total of 15 pseudochromosomes were constructed and accounted for 94.45% of the total genome length ( Table 4 ). This number (n = 15) is in agreement with the established karyotype for G. bimaculatus (n = 14 autosomes + X), which was previously determined by cytogenetic analysis ( Yoshimura et al., 2006 ) and is also consistent with that of the recently reported species G. assimilis ( Ito et al., 2025 ), which is closely related to G. bimaculatus . The X chromosome was identified as the longest, accounting for 15.68% of the genome, which is in accordance with the karyotype of this species ( Yoshimura et al., 2006 ). This was further validated by the observation that the X chromosome displayed half of the read coverage compared to the autosomal chromosomes, calculated using a genomic short-read library (DRR272308) from a single hemizygous male (Fig. S1). View this table: View inline View popup Table 4. Statistics of the assembled pseudochromosomes in G. bimaculatus . Gene annotation The final consensus gene set, derived from merging BRAKER ab initio predictions and StringTie2 RNA-seq-based predictions, comprised 14,964 protein-coding genes ( Table 5 ). This total is less than the 17,871 genes reported previously ( Ylla et al. 2021 ), likely because the improved scaffold contiguity of our assembly allows for the correct assembly of genes previously split across multiple scaffolds. Of the 14,964 predicted genes, functional annotation was assigned using eggNOG-mapper and BLASTp (E-value < 1.0 ×10 −10 against several model organisms’ annotation datasets. eggNOG-mapper annotated 71.96% of the genes. The BLASTp searches (E-value < 1.0 × 10 −10 ) against databases including H. sapiens, M. musculus, C. elegans, D. melanogaster , and UniProt Swiss-Prot yielded hit rates ranging from 44.84% to 71.89% ( Table 6 ). View this table: View inline View popup Download powerpoint Table 5. Summary statistics for gene prediction in G. bimaculatus . View this table: View inline View popup Table 6. Structural statistics of the G. bimaculatus gene set. The completeness of this annotation set was validated using BUSCO v6.0.0. The analysis identified 95.7% of the expected complete insecta BUSCOs (C:95.7%[S:93.6%,D:2.0%],F:1.5%,M:2.9%) ( Table 7 ). These results collectively indicate a high-quality gene set for this species. View this table: View inline View popup Download powerpoint Table 7. Functional annotation summary for the G. bimaculatus gene set. Recovery of missing neuropeptide genes A recent comprehensive study ( Mochizuki et al., 2023 ) highlighted significant gaps in the previous draft genome assembly ( Ylla et al., 2021 ). They reported that several crucial neuropeptides genes (e.g., ACP (Adipokinetic hormone/corazonin-related peptide), Allatotropin, Kinin, etc.) were missing from the draft genome and could only be identified within de novo transcriptome assemblies, suggesting these loci were absent from the previous reference. We asked if our new chromosome-scale assembly was more complete in this regard, by mapping the cDNA sequences of these previously missing neuropeptides. We successfully located all nine of these genes encoding neuropeptides (i.e., ACP, Allatostatin CC (Ast CC), Allatotropin, CCHamide-1, CCHamide-2, CRF/DH (Corticotropin releasing factor-like diuretic hormone), Kinin (Leucokinin), Neuropeptide F1a (NPF1a), Neuropeptide F1b (NPF1b)), which are now correctly anchored onto our pseudomolecules. For example, the ACP gene, previously missing, was successfully mapped to Chromosome X, where it spans 11,668 bp and is composed of three exons, revealing its complete exon-intron structure (Fig. S2). This demonstrates that our assembly not only improves contiguity to the chromosome scale but also recovers functionally critical genes that were absent in the previous reference, providing a more complete and reliable resource for functional genomics in G. bimaculatus . Conclusions We have generated a high-quality, chromosome-scale genome assembly and updated gene annotation for the key hemimetabolous model organism G. bimaculatus . This assembly represents a substantial upgrade to the previous draft sequence ( Ylla et al., 2021 ), increasing the scaffold N50 from 6.3 Mbp to 107.4 Mbp and anchoring 94.45% of the sequence into 15 pseudomolecules, consistent with the known karyotype ( Yoshimura et al., 2006 ). Crucially, our assembly resolves significant gaps present in the previous version, evidenced by the recovery of nine essential neuropeptide genes previously reported as missing ( Mochizuki et al., 2023 ). This improved completeness is further supported by superior BUSCO scores for both the genome (98.1% vs. 96.0%) and the gene set (95.7% vs. 81.2%). This highly contiguous and complete genome sequence provides an essential new foundation for the G. bimaculatus research community, facilitating advanced genetic and genomic analyses, such as synteny comparisons, QTL mapping, and the precise design of genome-editing experiments. Data Availability The scripts used for the analyses in this study are available in GitHub ( https://github.com/Kataoka-K-Lab/Gryllus_bimaculatus_genome_gbim_v2.2 ). All bioinformatics tools used in this study followed their respective manuals and protocols. The software versions, codes, and parameters are provided in the Materials and Methods section. Unless otherwise specified, default parameters were used. The genomic WGS sequencing data were deposited in the NCBI Sequence Read Archive (SRA) database under the BioProject PRJNA1347939. The assembled genome and annotation datasets are also available in figshare ( https://doi.org/10.6084/m9.figshare.30472754.v1 ) ( Kataoka, 2025 ). Funding This study was supported by the Cabinet Office, Government of Japan, Cross-ministerial Moonshot Agriculture, Forestry and Fisheries Research and Development Program, “Technologies for Smart Bio-industry and Agriculture” (BRAIN) [JPJ009237] (K.K., S.I., T.A., K.Y., and T.M.), the Strategic Programme Excellence Initiative at the Jagiellonian University – BioS PRA (T.G. and G.Y.), and the National Science Foundation Award [IOS-2220747] (C.G.E.). C.G.E. is an investigator of the Howard Hughes Medical Institute. Competing Interests The authors declare no conflict of interest. Supplementary Figures Supplementary Figure S1. Genomic read coverage confirms X chromosome hemizygosity . Mean read coverage depth across all assembled chromosomes derived from the mapping of genomic short reads of a single male Gryllus bimaculatus individual (SRA: DRR272308). The X chromosome displays approximately half the average autosomal coverage depth, consistent with the expected pattern of male hemizygosity. Supplementary Figure S2. Recovery of the Adipokinetic hormone/corazonin-related peptide (ACP) gene, previously missing from the draft genome . The image displays an Integrative Genomics Viewer (IGV) screenshot of the G. bimaculatus chromosome-scale assembly. The complete gene model for ACP (Gbim.chrXG0009740.1), one of the nine neuropeptide genes reported missing from the first assembly report ( Mochizuki et al., 2023 ), is shown. The gene is now successfully anchored and annotated on Chromosome X (chrX), spanning a region of approximately 20 kb. The blue track (Gbim_v2.2.gff3) shows the full exon-intron structure (exons as thick blocks, introns as thin lines). Funder Information Declared Technologies for Smart Bio-industry and Agriculture , JPJ009237 Strategic Programme Excellence Initiative at the Jagiellonian University – BioS PRA National Science Foundation Award , IOS-2220747 References ↵ Abe , T. , Tada , C. , & Nagayama , T. ( 2021 ). Winner and loser effects of juvenile cricket Gryllus bimaculatus . Journal of Ethology , 39 ( 1 ), 47 – 54 . OpenUrl ↵ Bonenfant , Q. , Noé , L. , & Touzet , H. ( 2023 ). Porechop_ABI: discovering unknown adapters in Oxford Nanopore Technology sequencing reads for downstream trimming . Bioinformatics Advances , 3 ( 1 ), vbac085 . OpenUrl ↵ Brůna , T. , Hoff , K. J. , Lomsadze , A. , Stanke , M. , & Borodovsky , M. ( 2021 ). BRAKER2: automatic eukaryotic genome annotation with GeneMark-EP+ and AUGUSTUS supported by a protein database . NAR Genomics and Bioinformatics , 3 ( 1 ), qaa108. ↵ Cantalapiedra , C. P. , Hernández-Plaza , A. , Letunic , I. , Bork , P. , & Huerta-Cepas , J. ( 2021 ). eggNOG-mapper v2: Functional Annotation, Orthology Assignments, and Domain Prediction at the Metagenomic Scale . Molecular Biology and Evolution , 38 ( 12 ), 5825 – 5829 . OpenUrl CrossRef PubMed ↵ Challis , R. , Richards , E. , Rajan , J. , Cochrane , G. , & Blaxter , M. ( 2020 ). BlobToolKit - interactive quality assessment of genome assemblies . G3 , 10 ( 4 ), 1361 – 1374 . OpenUrl Abstract / FREE Full Text ↵ Dainat , J. , Hereñú , D. , Davis , E. , Crouch , K. , Lucile Sol , Pascal-Git, & Tayyrov. ( 2022 ). NBISweden/AGAT: AGAT-v0.9.1 . Zenodo . doi: 10.5281/ZENODO.6488306 OpenUrl CrossRef ↵ Donoughe , S. , & Extavour , C. G. ( 2016 ). Embryonic development of the cricket Gryllus bimaculatus . Developmental Biology , 411 ( 1 ), 140 – 156 . OpenUrl CrossRef PubMed ↵ Dossey , A. T. , Oppert , B. , Chu , F.-C. , Lorenzen , M. D. , Scheffler , B. , Simpson , S. , Koren , S. , Johnston , J. S. , Kataoka , K. , & Ide , K. ( 2023 ). Genome and genetic engineering of the house cricket (Acheta domesticus): A resource for sustainable agriculture . Biomolecules , 13 ( 4 ), 589 . OpenUrl CrossRef PubMed ↵ Dudchenko , O. , Batra , S. S. , Omer , A. D. , Nyquist , S. K. , Hoeger , M. , Durand , N. C. , Shamim , M. S. , Machol , I. , Lander , E. S. , Aiden , A. P. , & Aiden , E. L. ( 2017 ). De novo assembly of the Aedes aegypti genome using Hi-C yields chromosome-length scaffolds . Science , 356 ( 6333 ), 92 – 95 . OpenUrl Abstract / FREE Full Text ↵ Durand , N. C. , Shamim , M. S. , Machol , I. , Rao , S. S. P. , Huntley , M. H. , Lander , E. S. , & Aiden , E. L. ( 2016 ). Juicer provides a one-click system for analyzing loop-resolution Hi-C experiments . Cell Systems , 3 ( 1 ), 95 – 98 . OpenUrl PubMed ↵ Flynn , J. M. , Hubley , R. , Goubert , C. , Rosen , J. , Clark , A. G. , Feschotte , C. , & Smit , A. F. ( 2020 ). RepeatModeler2 for automated genomic discovery of transposable element families . Proceedings of the National Academy of Sciences of the United States of America , 117 ( 17 ), 9451 – 9457 . OpenUrl Abstract / FREE Full Text ↵ Gabriel , L. , Brůna , T. , Hoff , K. J. , Ebel , M. , Lomsadze , A. , Borodovsky , M. , & Stanke , M. ( 2024 ). BRAKER3: Fully automated genome annotation using RNA-seq and protein evidence with GeneMark-ETP, AUGUSTUS, and TSEBRA . Genome Research , 34 ( 5 ), 769 – 777 . OpenUrl Abstract / FREE Full Text ↵ Haas , B. J. , Papanicolaou , A. , Yassour , M. , Grabherr , M. , Blood , P. D. , Bowden , J. , Couger , M. B. , Eccles , D. , Li , B. , Lieber , M. , MacManes , M. D. , Ott , M. , Orvis , J. , Pochet , N. , Strozzi , F. , Weeks , N. , Westerman , R. , William , T. , Dewey , C. N. , … Regev , A. ( 2013 ). De novo transcript sequence reconstruction from RNA-seq using the Trinity platform for reference generation and analysis . Nature Protocols , 8 ( 8 ), 1494 – 1512 . OpenUrl PubMed ↵ Hoff , K. J. , Lange , S. , Lomsadze , A. , Borodovsky , M. , & Stanke , M. ( 2016 ). BRAKER1: Unsupervised RNA-Seq-based genome annotation with GeneMark-ET and AUGUSTUS . Bioinformatics , 32 ( 5 ), 767 – 769 . OpenUrl CrossRef PubMed ↵ Hoff , K. J. , Lomsadze , A. , Borodovsky , M. , & Stanke , M. ( 2019 ). Whole-genome annotation with BRAKER . Methods in Molecular Biology (Clifton, N.J .), 1962 , 65 – 95 . OpenUrl CrossRef PubMed ↵ Horch , H. W. , Mito , T. , Popadic , A. , Ohuchi , H. , & Noji , S. (Eds.). ( 2017 ). The cricket as a model organism: Development, regeneration, and behavior . Springer . ↵ Ito , Y. , Sanno , R. , Ashikari , S. , Yura , K. , Asahi , T. , Ylla , G. , & Kataoka , K. ( 2025 ). Chromosome-scale whole genome assembly and annotation of the Jamaican field cricket Gryllus assimilis . Scientific Data , 12 ( 1 ), 826 . OpenUrl PubMed ↵ Kataoka , K. ( 2025 ). Chromosome-scale genome assembly and annotation of the two-spotted cricket Gryllus bimaculatus (Orthoptera: Gryllidae) [Data set] . figshare . doi: 10.6084/M9.FIGSHARE.30472754.V1 OpenUrl CrossRef ↵ Kataoka , K. , Minei , R. , Ide , K. , Ogura , A. , Takeyama , H. , Takeda , M. , Suzuki , T. , Yura , K. , & Asahi , T. ( 2020 ). The draft genome dataset of the Asian cricket Teleogryllus occipitalis for molecular research toward entomophagy . Frontiers in Genetics , 11 , 470 . OpenUrl PubMed ↵ Kataoka , K. , Togawa , Y. , Sanno , R. , Asahi , T. , & Yura , K. ( 2022 ). Dissecting cricket genomes for the advancement of entomology and entomophagy . Biophysical Reviews , 14 ( 1 ), 75 – 97 . OpenUrl PubMed ↵ Kim , D. , Paggi , J. M. , Park , C. , Bennett , C. , & Salzberg , S. L. ( 2019 ). Graph-based genome alignment and genotyping with HISAT2 and HISAT-genotype . Nature Biotechnology , 37 ( 8 ), 907 – 915 . OpenUrl CrossRef PubMed ↵ Kolmogorov , M. , Yuan , J. , Lin , Y. , & Pevzner , P. A. ( 2019 ). Assembly of long, error-prone reads using repeat graphs . Nature Biotechnology , 37 ( 5 ), 540 – 546 . OpenUrl CrossRef PubMed ↵ Kovaka , S. , Zimin , A. V. , Pertea , G. M. , Razaghi , R. , Salzberg , S. L. , & Pertea , M. ( 2019 ). Transcriptome assembly from long-read RNA-seq alignments with StringTie2 . Genome Biology , 20 ( 1 ), 278 . OpenUrl CrossRef PubMed ↵ Kuriwada , T. ( 2022 ). Encounter with heavier females changes courtship and fighting efforts of male field crickets Gryllus bimaculatus (Orthoptera: Gryllidae) . Journal of Ethology , 40 ( 2 ), 145 – 151 . OpenUrl ↵ Matsumoto , Y. , Matsumoto , C. S. , & Mizunami , M. ( 2018 ). Signaling Pathways for Long-Term Memory Formation in the Cricket . Frontiers in Psychology , 9 , 1014 . OpenUrl PubMed ↵ Matsuoka , Y. , Nakamura , T. , Watanabe , T. , Barnett , A. A. , Tomonari , S. , Ylla , G. , Whittle , C. A. , Noji , S. , Mito , T. , & Extavour , C. G. ( 2025 ). Establishment of CRISPR/Cas9-based knock-in in a hemimetabolous insect: targeted gene tagging in the cricket Gryllus bimaculatus . Development , 152 ( 1 ), dev199746 . OpenUrl PubMed ↵ Mito , T. , Ishimaru , Y. , Watanabe , T. , Nakamura , T. , Ylla , G. , Noji , S. , & Extavour , C. G. ( 2022 ). Cricket: The third domesticated insect . Current Topics in Developmental Biology , 147 , 291 – 306 . OpenUrl CrossRef PubMed ↵ Mito , T. , & Noji , S. ( 2008 ). The two-spotted cricket Gryllus bimaculatus: An emerging model for developmental and regeneration studies . Cold Spring Harbor Protocols , 2008 ( 12 ), db.emo110. ↵ Miyawaki , K. , Mito , T. , Sarashina , I. , Zhang , H. , Shinmyo , Y. , Ohuchi , H. , & Noji , S. ( 2004 ). Involvement of Wingless/Armadillo signaling in the posterior sequential segmentation in the cricket, Gryllus bimaculatus (Orthoptera), as revealed by RNAi analysis . Mechanisms of Development , 121 ( 2 ), 119 – 130 . OpenUrl CrossRef PubMed Web of Science ↵ Mochizuki , T. , Sakamoto , M. , Tanizawa , Y. , Seike , H. , Zhu , Z. , Zhou , Y. J. , Fukumura , K. , Nagata , S. , & Nakamura , Y. ( 2023 ). Best Practices for Comprehensive Annotation of Neuropeptides of Gryllus bimaculatus . Insects , 14 ( 2 ), 121 . OpenUrl PubMed ↵ Nakamura , T. , Mito , T. , Bando , T. , Ohuchi , H. , & Noji , S. ( 2008 ). Dissecting insect leg regeneration through RNA interference: Dissecting insect leg regeneration through RNA interference . Cellular and Molecular Life Sciences , 65 ( 1 ), 64 – 72 . OpenUrl CrossRef PubMed ↵ Pertea , G. , & Pertea , M. ( 2020 ). GFF utilities: GffRead and GffCompare . F1000Research , 9 ( 304 ), 304 . OpenUrl ↵ Slater , G. S. C. , & Birney , E. ( 2005 ). Automated generation of heuristics for biological sequence comparison . BMC Bioinformatics , 6 ( 1 ), 31 . OpenUrl CrossRef PubMed ↵ Smit , A. F. A. , Hubley , R. , & Green , P. ( 2015 ). RepeatMasker (Open-4.0) [Computer software] . ↵ Stanke , M. , Diekhans , M. , Baertsch , R. , & Haussler , D. ( 2008 ). Using native and syntenically mapped cDNA alignments to improve de novo gene finding . Bioinformatics , 24 ( 5 ), 637 – 644 . OpenUrl CrossRef PubMed Web of Science ↵ Stanke , M. , Schöffmann , O. , Morgenstern , B. , & Waack , S. ( 2006 ). Gene prediction in eukaryotes with a generalized hidden Markov model that uses hints from external sources . BMC Bioinformatics , 7 ( 1 ), 62 . OpenUrl CrossRef PubMed ↵ Szrajer , S. , Gray , D. , & Ylla , G. ( 2024 ). The genome assembly and annotation of the cricket Gryllus longicercus . Scientific Data , 11 ( 1 ), 708 . OpenUrl PubMed ↵ Tegenfeldt , F. , Kuznetsov , D. , Manni , M. , Berkeley , M. , Zdobnov , E. M. , & Kriventseva , E. V. ( 2025 ). OrthoDB and BUSCO update: annotation of orthologs with wider sampling of genomes . Nucleic Acids Research , 53 ( D1 ), D516 – D522 . OpenUrl CrossRef PubMed ↵ Ylla , G. , Nakamura , T. , Itoh , T. , Kajitani , R. , Toyoda , A. , Tomonari , S. , Bando , T. , Ishimaru , Y. , Watanabe , T. , Fuketa , M. , Matsuoka , Y. , Barnett , A. A. , Noji , S. , Mito , T. , & Extavour , C.G. ( 2021 ). Insights into the genomic evolution of insects from cricket genomes . Communications Biology , 4 ( 1 ), 733 . OpenUrl PubMed ↵ Yoshimura , A. , Nakata , A. , Mito , T. , & Noji , S. ( 2006 ). The characteristics of karyotype and telomeric satellite DNA sequences in the cricket, Gryllus bimaculatus (Orthoptera, Gryllidae) . Cytogenetic and Genome Research , 112 ( 3–4 ), 329 – 336 . OpenUrl PubMed ↵ Zimin , A. V. , & Salzberg , S. L. ( 2020 ). The genome polishing tool POLCA makes fast and accurate corrections in genome assemblies . PLoS Computational Biology , 16 ( 6 ), e1007981 . OpenUrl View the discussion thread. Back to top Previous Next Posted November 03, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Chromosome-scale genome assembly and annotation of the two-spotted cricket Gryllus bimaculatus (Orthoptera: Gryllidae) Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Chromosome-scale genome assembly and annotation of the two-spotted cricket Gryllus bimaculatus (Orthoptera: Gryllidae) Kosuke Kataoka , Ryuto Sanno , Tomasz Gaczorek , Upendra Raj Bhattarai , Yuki Ito , Shintaro Inoue , Kei Yura , Toru Asahi , Guillem Ylla , Taro Mito , Cassandra G. Extavour bioRxiv 2025.10.31.685973; doi: https://doi.org/10.1101/2025.10.31.685973 Share This Article: Copy Citation Tools Chromosome-scale genome assembly and annotation of the two-spotted cricket Gryllus bimaculatus (Orthoptera: Gryllidae) Kosuke Kataoka , Ryuto Sanno , Tomasz Gaczorek , Upendra Raj Bhattarai , Yuki Ito , Shintaro Inoue , Kei Yura , Toru Asahi , Guillem Ylla , Taro Mito , Cassandra G. Extavour bioRxiv 2025.10.31.685973; doi: https://doi.org/10.1101/2025.10.31.685973 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genomics Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17690) Bioengineering (13892) Bioinformatics (41936) Biophysics (21451) Cancer Biology (18588) Cell Biology (25499) Clinical Trials (138) Developmental Biology (13378) Ecology (19899) Epidemiology (2067) Evolutionary Biology (24320) Genetics (15609) Genomics (22506) Immunology (17736) Microbiology (40394) Molecular Biology (17181) Neuroscience (88603) Paleontology (666) Pathology (2832) Pharmacology and Toxicology (4824) Physiology (7641) Plant Biology (15152) Scientific Communication and Education (2045) Synthetic Biology (4294) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00