Full text
36,507 characters
· extracted from
preprint-html
· click to expand
Chromosome-scale genome assembly of acerola (Malpighia emarginata DC.) | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Chromosome-scale genome assembly of acerola ( Malpighia emarginata DC.) View ORCID Profile Kenta Shirasawa , Kazuhiko Harada , Noriaki Haramoto , Hitoshi Aoki , Shota Kammera , Masashi Yamamoto , Yu Nishizawa doi: https://doi.org/10.1101/2024.07.30.605750 Kenta Shirasawa 1 Kazusa DNA Research Institute , Chiba 292-0818, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kenta Shirasawa For correspondence: shirasaw{at}kazusa.or.jp Kazuhiko Harada 2 Nichirei Foods Inc. , Chiba 261-0002, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Noriaki Haramoto 2 Nichirei Foods Inc. , Chiba 261-0002, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hitoshi Aoki 2 Nichirei Foods Inc. , Chiba 261-0002, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Shota Kammera 3 Faculty of Agriculture, Kagoshima University , Kagoshima 890-0065, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Masashi Yamamoto 3 Faculty of Agriculture, Kagoshima University , Kagoshima 890-0065, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yu Nishizawa 3 Faculty of Agriculture, Kagoshima University , Kagoshima 890-0065, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Acerola ( Malpighia emarginata DC.) is a tropical evergreen shrub that produces vitamin C-rich fruits. Increasing fruit nutrition is one of the main targets of acerola breeding programs. Genomic tools have been shown to accelerate plant breeding even in fruiting tree species, which generally have a long life cycle; however, the availability of genomic resources in acerola, so far, has been limited. In this study, as a first step toward developing an efficient breeding technology for acerola, we established a chromosome-scale genome assembly of acerola using high-fidelity long-read sequencing and genetic mapping. The resultant assembly comprises 10 chromosome-scale sequences that span a physical distance of 1,032.5 Mb and contain 35,892 predicted genes. Phylogenetic analysis of genome-wide SNPs in 60 acerola breeding materials revealed three distinct genetic groups. Overall, the genomic resource of acerola developed in this study, including its genome and gene sequences, genetic map, and phylogenetic relationship among breeding materials, will not only be useful for acerola breeding but will also facilitate genomic and genetic studies on acerola and related species. Introduction Acerola ( Malpighia emarginata DC.) is a tropical, fruit-bearing, evergreen shrub belonging to the family Malpighiaceae in the order Malpighiales. Acerola, also known as Barbados cherry and West Indian cherry, is native to Central and South America and the Caribbean Islands 1 . Acerola is cultivated in tropical and subtropical countries, especially in northwestern Brazil and the Mekong Delta region of Vietnam (Tien Giang Province and Ben Tre Province), primarily to meet the demand of the fresh fruit market and/or processing industries 2 . Because acerola fruits are rich in vitamin C (ascorbic acid), their concentrated juice is used as a raw material for preparing beverages and supplemental vitamin C powder. Therefore, one of the targets of acerola breeding programs is to increase the vitamin C content of acerola fruits. However, acerola, like other fruit tree species, has a long life cycle, making its breeding using conventional methods challenging. Genomics can accelerate breeding programs not only in cereal and vegetable crops but also in fruit trees 3 . Indeed, novel genomics-based approaches, e.g., genome-wide association study and genomic selection based on machine learning methods, Bayesian networks, image analysis, and genomic prediction, have been implemented in the breeding programs of citrus, apple, and Japanese pear 4 – 6 . While massive genome information and multi-omics profiles have been available for the above-mentioned fruit trees 4 , 7 , the publicly available genome data of acerola, besides its chromosome number (2n = 20) 8 , has been limited. This situation has caused the breeding of acerola to lag behind that of other fruit crops. High-fidelity long-read sequencing, also known as HiFi (PacBio) sequencing, has enabled genome assembly at the chromosome level or telomere-to-telomere level in several species 9 . The telomere-to-telomere assembly comprises a single contig, with telomere repeats at both ends, that covers the entire chromosome without any gaps of undetermined nucleotide sequences. Sequence scaffolding methods, together with the HiFi technology, help to establish chromosome-level genome assembly. Hi-C is a popular method used to connect the assembled sequences into the chromosome-level assembly 10 . Alternatively, genetic mapping with high-density genetic loci, based on a classical Mendelian law, enables anchoring the genome sequence fragments to linkage maps for establishing chromosome-level sequences 11 . The establishment of genome resources and genetic diversity assessment of breeding materials would be the first step in the breeding of acerola. In this study, we used HiFi technology to sequence the acerola genome and then anchored the sequences to a genetic map to construct a chromosome-level genome assembly of acerola. Subsequently, we assessed the genetic diversity of acerola breeding materials using SNPs derived by double digest restriction-site associated sequencing (ddRAD-Seq). The acerola genome information generated in this study will be helpful in accelerating acerola breeding programs and increasing knowledge of the genetics and genomics of Malpighiales. Materials and Methods Plant materials Acerola cultivar NRA309, which was registered in 2008 by Nichirei Foods Inc. (Tokyo, Japan), was used for genome sequencing. An S1 mapping population (nC=C118), derived by the self-pollination of NRA309, was used to construct a genetic map of acerola (Supplementary Table S1). In addition to NRA309, a total of 59 acerola lines were used for genetic diversity analysis. These 59 lines included 10 lines from Brazil, four from Japan, and nine from USA as well as 36 progenies generated in breeding programs by crossing the above-mentioned lines (Supplementary Table S2). All plant materials were maintained at the Faculty of Agriculture, Kagoshima University. Genome size estimation Genomic DNA (gDNA) of NRA309 was extracted from leaves using Genomic Tip (Qiagen, Hilden, Germany). The PCR-free Swift 2S Turbo Flexible DNA Library Kit (Swift Sciences, Ann Arbor, MI, USA) was used to construct a short-read library, which was then converted into a DNA nanoball sequencing library using the MGI Easy Universal Library Conversion Kit (MGI Tech, Shenzhen, China). The library was sequenced using a DNBSEQ G400 instrument (MGI Tech) to generate 100-bp paired-end reads. The genome size of NRA309 was estimated using Jellyfish ( k -mer size = 17) 12 . Genome sequencing and primary assembly NRA309 gDNA was subjected to long-read library preparation. Briefly, the NRA309 gDNA was sheared to an average fragment size of 40 kb using Megaruptor 2 (Deagenode, Liege, Belgium) in the Large Fragment Hydropore mode, and the sheared DNA was used for HiFi SMRTbell library preparation with the SMRTbell Express Template Prep Kit 2.0 (PacBio, Menlo Park, CA, USA). The obtained DNA libraries were fractionated with BluePippin (Sage Science, Beverly, MA, USA) to eliminate fragments shorter than 20 kb in length. The fractionated DNA libraries were sequenced on a SMRT Cell 8M on the Sequel II system (PacBio). HiFi reads were constructed using the CCS pipeline ( https://ccs.how ) and assembled using Hifiasm (version 0.16.1) 13 with default parameters. Organelle genome sequences, identified by sequence similarity searches of the reported plastid and mitochondrial genome sequences of acerola relatives (Supplementary Table S3) using Minimap2 (version 2.24) 14 , were eliminated. Assembly completeness was evaluated with the embryophyta_odb10 data using Benchmarking Universal Single-Copy Orthologs (BUSCO) (version 5.2.2) 15 . Chromosome-level genome assembly by genetic mapping The gDNA of lines in the S1 population was extracted from leaves using the Plant Genomic DNA Extraction Mini Kit (Favorgen, Ping-Tung, Taiwan). The extracted gDNA was digested with Pst I and Msp I restriction endonucleases and subjected to ddRAD-Seq library preparation 16 . The resultant library was sequenced on a DNBSEQ G400 instrument (MGI Tech) to generate 100-bp paired-end reads. After removing adapter sequences (AGATCGGAAGAGC) with fastx_clipper in the FASTX-Toolkit (version 0.0.14, http://hannonlab.cshl.edu/fastx_toolkit ) and trimming low-quality reads (quality score < 10) with PRINSEQ (version 0.20.4) 17 , high-quality reads were aligned onto the genome assembly of NRA309 using Bowtie2 (version 2.3.5.1) 18 . High-confidence biallelic SNPs were identified using the mpileup and call options of BCFtools (version 1.9) 19 and filtered using VCFtools (version 0.1.16) 20 based on the following criteria: read depth ≥ 5; SNP quality = 999; minor allele frequency ≥ 0.2; proportion of missing data < 20%. A linkage analysis of the SNPs was performed using Lep-Map3 (version 0.2) 21 to construct a genetic map. Contig sequences were anchored to the genetic map, and chromosome-level pseudomolecule sequences were established using ALLMAPS (version 0.7.3) 11 . Telomere sequences containing repeats of a 7-bp motif (5C-TTTAGGG-3C) were searched using the search subcommand of tidk ( https://github.com/tolkit/telomeric-identifier ), with a window size of 100,000 bp. Prediction of protein-coding genes and repetitive sequences Protein-coding genes were predicted using the ab initio strategy of BRAKER3 (version 3.0.7) 22 and Helixer (version 0.3.2) 23 . Prediction completeness was evaluated with the embryophyta_odb10 data using BUSCO (version 5.2.2) 15 . The predicted genes were functionally annotated using emapper (version 2.1.12) implemented in EggNOG 24 , in conjunction with searches against the UniProtKB database 25 and Arabidopsis peptide sequences (Araport11) 26 using DIAMOND (version 2.0.14) 27 and BLAST 28 , respectively. Repetitive sequences in the assembly were identified using RepeatMasker ( https://www.repeatmasker.org ), based on repeat sequences registered in Repbase and a de novo repeat library built with RepeatModeler ( https://www.repeatmasker.org ). Genetic diversity analysis The genome structure of acerola was compared with that of three Malpighiales species, including rubber tree ( Hevea brasiliensis , NCBI RefSeq assembly GCF_030052815.1) 29 , cassava ( Manihot esculenta , GCF_001659605.2) 30 , and castor bean ( Ricinus communis , GCF_019578655.1) 31 , using Minimap2 14 implemented in D-Genies 32 . ddRAD-Seq libraries of all 59 acerola lines were prepared as described above and sequenced on a DNBSEQ G400 instrument (MGI Tech). Subsequently, the ddRAD-seq reads obtained from all 59 lines and NRA309 were aligned onto the chromosome-level pseudomolecule sequences to detect sequence variants. High-confidence biallelic SNPs were selected with the filtering conditions: read depth ≥ 5; SNP quality = 999; minor allele frequency ≥ 0.05; proportion of missing data < 20%. The population structure of 60 lines was evaluated based on the maximum-likelihood estimation of individual ancestries using ADMIXTURE (version 1.3.0) 33 . A phylogenetic tree, based on 100 bootstrap replicates, was created with SNPhylo (version 20140701) 34 and visualized with iTOL (version 6.9.1) 35 . Results Genome sequencing and assembly To estimate the acerola genome size, clean short-read data (73.1 Gb) were subjected to k -mer distribution analysis. The results indicated that the acerola genome was highly heterozygous, with a haploid genome size of 1,114.2CMb (Supplementary Figure S1). A total of 90.3 million HiFi reads, with an N50 length of 32.4 kb (28.3 Gb; 24.6× coverage of the estimated genome size), were obtained from one SMRT Cell 8M and assembled into 73 contigs. Three potential contaminant sequences (221.1 kb in total) from organelle genomes were removed. Consequently, 70 contigs with a total length of 1,053.0 Mb and N50 and N90 lengths of 73.1 Mb and 39.1 Mb, respectively ( Table 1 ), were obtained. These 70 contigs represented the genome assembly of acerola and were designated as NRA309_r1.0. A genetic map of acerola was constructed based on 940.0 million ddRAD-Seq reads obtained from 118 S1 lines and NRA309 (parental line). Briefly, high-quality reads were aligned to the assembled contigs as a reference, with an average mapping rate of 88.6% (Supplementary Table S1), and 7,935 high-confidence SNPs were detected. In the subsequent linkage analysis, 10 linkage groups were obtained, with the number of linkage groups corresponding to the basic chromosome number of acerola. The resultant genetic map spanned a distance of 1,059.8 cM and contained a total of 7,775 SNPs ( Table 2 , Supplementary Figure S2). Contigs were aligned onto the acerola genetic map to establish chromosome-level sequences. Of the 70 contigs, 19 were mapped to 10 linkage groups. Among them, 1, 2, and 3 contigs were mapped on 3, 5, and 3 linkage groups, respectively ( Table 2 ). The 10 pseudomolecule sequences spanned a physical distance of 1,032.5 Mb (98.1% of the total contig length) ( Tables 1 , 2 , Supplementary Figure S2). Telomere repeats were found at one end of seven sequences and at both ends of two sequences, while no telomeres were found in one sequence ( Table 2 ). Overall, chromosomes 5 and 7 were completely assembled, without gaps, at the telomere-to-telomere level. The remaining 51 contigs (total length = 20.4 Mb) with five telomere repeat units were not assigned to any linkage groups ( Table 1 , 2). In accordance with the BUSCO score, the acerola genome assembly was 98.8% complete ( Table 3 ). View this table: View inline View popup Download powerpoint Table 1 Statistics of the acerola genome assembly View this table: View inline View popup Download powerpoint Table 2 Statistics of the acerola pseudomolecule sequence View this table: View inline View popup Download powerpoint Table 3 Completeness of the genome assembly and predicted genes of acerola Gene and repetitive sequence prediction Next, we predicted protein-coding genes in the acerola genome. First, a total of 196,853 genes, with an average coding sequence length of 1,617 bp, were predicted using BRAKER3. However, the BUSCO score of these genes was only 82.5% (41.1% single-copy complete BUSCOs and 41.1% duplicated complete BUSCOs). Because of the high gene number and low BUSCO score, we speculated that BRAKER3-based prediction was inaccurate. Therefore, we repeated the gene prediction using Helixer. As a result, a total of 35,892 genes (average coding sequence length = 1,150 bp) were predicted, of which 35,551 and 341 genes were located on the 10 chromosome-scale sequences and the 51 unassigned contigs, respectively. The BUSCO score of Helixer-predicted genes was 99.1% ( Table 3 ). Out of the 35,892 genes, 31,742 (88.4%), 30,989 (86.3%), and 29,068 (81.0%) genes were functionally annotated with EggNOG mapper, DIAMOND search against UniProtKB, and BLAST search against Araport11 peptide sequences, respectively (Supplementary Table S4). In total, 32,239 (89.8%) genes were annotated by at least one of the three methods. Repetitive sequences occupied a total of 816.3CMb (77.5%) of the pseudomolecule sequences. Nine major types of repeats were identified in varying proportions ( Table 4 ). LTR retroelements (427.2CMb) represented the dominant repeat type in the pseudomolecule sequences, followed by DNA transposons (89.8CMb). Repeat sequences unavailable in public databases totaled 252.5CMb. View this table: View inline View popup Download powerpoint Table 4 Repetitive sequences in the acerola genome Comparative genome structure analysis The pseudomolecule sequences of acerola were aligned against those of rubber tree ( H. brasiliensis ), cassava ( M. esculenta ), and castor bean ( R. communis ), all of which are members of Malpighiaceae. Unexpectedly, no clear similarity was detected, and probable conserved structures were highly fragmented between the genome of acerola and those of the other three Malpighiaceae species (Supplementary Figure S3). Genetic diversity of acerola lines To evaluate the genetic diversity of acerola lines, the gDNA of 59 lines was subjected to ddRAD-Seq. An average of 12.7 million ddRAD-Seq reads per sample were obtained. High-quality reads from 59 lines and NRA309 were aligned to the pseudomolecule sequences of NRA309 (reference), with an average mapping rate of 90.6% (Supplementary Table S2), and 49,070 high-confidence SNPs were detected. According to SnpEff results, the most prominent SNP type was modifier impact (59.5%) in introns and intergenic regions, followed by low impact (21.9%; synonymous mutations), moderate impact (18.2%; missense mutations), and high impact (0.4%; nonsense and splice-site mutations) (Supplementary Table S5). ADMIXTURE analysis grouped the 60 lines into three clusters (K = 3): cluster a, seven Japanese and Hawaiian lines; cluster b, 16 Hawaiian and Brazilian lines; and cluster c, 36 lines derived from crosses between FB05 and five members of cluster b ( Figure 1a ). These three clusters were well-supported by phylogenetic analysis ( Figure 1b ). Download figure Open in new tab Figure 1 Genetic structure and phylogenetic tree of 60 acerola lines. (a) Population structure of 60 acerola lines. Each color represents a distinct group. The origin of acerola lines is shown in parentheses: BRA, Brazil; JPN, Japan; USA, United States of America; and Cross, Progenies of crosses. (b) Phylogenetic tree. Numbers in red on branches indicate bootstrap values based on 100 replicates. Discussion Here, we present a chromosome-scale genome assembly of acerola, a member of the family Malpighiaceae. To the best of our knowledge, this is the first report of a chromosome-scale genome assembly of not only acerola but also a Malpighiaceae species. Chromosome-level genome assemblies have been reported for at least three species in the order Malpighiales, namely, rubber tree ( H. brasiliensis ), cassava ( M. esculenta ), and castor bean ( R. communis ), all of which are members of the family Euphorbiaceae. Comparative genome structure analysis revealed that the genome structures of acerola and the three Euphorbiaceae species were seldom conserved (Supplementary Figure S3). This suggests the possibility that acerola diverged from members of Euphorbiaceae within the order Malpighiales. More chromosome-level genome assemblies for members of the Malpighiaceae would be required to validate this hypothesis. The acerola genome information generated in this study would be useful for acerola breeding programs, in which the improvement of fruit vitamin C content together with plant disease and pest resistance are frequently targeted. The phylogenetic relationship of acerola cultivars determined in this study ( Figure 1 ) was in agreement with that determined previously using PCR-based sequence-related amplified polymorphism (SRAP) markers 36 . Parental lines in breeding programs could be adequately selected to enhance the genetic diversity of breeding materials. Furthermore, agronomically important loci could be identified through quantitative trait locus analysis, based on the genetic map. Once the genetic loci of interest have been determined, the underlying candidate genes could be easily identified using genome information. Being a woody plant, the life cycle or generation time of acerola is longer than that of annual herbaceous plants including vegetables, which makes the completion of conventional genetic analysis and breeding within a limited time period challenging. Therefore, the application of novel genomics-based approaches, e.g., genome-wide association study and genomic selection based on machine learning methods, Bayesian networks, image analysis, and genomic prediction 4 – 6 , has been proposed for fruit tree crops including woody plants. We conclude that the acerola genome resources developed in this study, including its genome and gene sequences, genetic map, and phylogenetic relationship among breeding materials, would be useful for genomics research on acerola and Malpighiales and for the development of novel fast-paced breeding technologies. This study provides a standard genome resource for the genomics, genetics, and breeding of acerola and related species. Data Availability Raw sequence reads were deposited in the DNA Data Bank of Japan (DDBJ) BioProject database under the accession number PRJDB18209, for which details are listed in Supplementary Tables S1 and S2. The assembled sequences are available at DDBJ ( accession numbers: AP035802 - AP035862 ) and Plant GARDEN ( https://plantgarden.jp ) 37 . Funding This work was supported by JSPS KAKENHI (22H05172 and 22H05181) and Kazusa DNA Research Institute Foundation. Conflict of Interest KH, NH, and HA are employees of Nichirei Foods Inc. All other authors declare no competing interests. Supplementary Figure S1 Estimated size of the acerola genome, based on k -mer analysis ( k = 17), with the given multiplicity values. Supplementary Figure S2 Genetic and physical maps of the acerola genome. Left: SNP loci on the genetic map (vertical lines) and physical map (bars) are connected with horizontal lines. Right: Positions of SNP loci are indicated with dots on the genetic map (y-axis, cM) and physical map (x-axis, Mb). Supplementary Figure S3 Comparative analysis of the genome sequence and structure of acerola and three other Malpighiales species. Dots indicate similarities in the genome structures of rubber tree ( Hevea brasiliensis ), cassava ( Manihot esculenta ), and castor bean ( Ricinus communis ). Chromosome numbers are indicated above the x-axis and on the right side of the y-axis, and genome sizes (Mb) are shown below the x-axis and on the left side of the y-axis. Acknowledgments We thank Y. Kishida, C. Minami, K. Ozawa, H. Tsuruoka, and A. Watanabe (Kazusa DNA Research Institute) for technical assistance. References 1. ↵ Vilvert , J. C. , Freitas , S. T. de, Veloso , C. M. , and Amaral , C. L. F . 2024 , Genetic Diversity on Acerola Quality: A Systematic Review . Braz. Arch. Biol. Technol ., 67 , e24220490 . OpenUrl 2. ↵ Ferreira , M. A. R. , Vilvert , J. C. , da Silva , B. O. S. , Ferreira , I. C. , de França Souza , F. , and de Freitas , S. T. 2022 , Multivariate selection index of acerola genotypes for fresh consumption based on fruit physicochemical attributes . Euphytica , 218 , 25 . OpenUrl 3. ↵ Varshney , R. K. , Bohra , A. , Yu , J. , Graner , A. , Zhang , Q. , and Sorrells , M. E . 2021 , Designing Future Crops: Genomics-Assisted Breeding Comes of Age . Trends Plant Sci ., 26 , 631 – 49 . OpenUrl 4. ↵ Iwata , H. , Minamikawa , M. F. , Kajiya-Kanegae , H. , Ishimori , M. , and Hayashi , T . 2016 , Genomics-assisted breeding in fruit trees . Breed. Sci ., 66 , 100 – 15 . OpenUrl 5. Iwata , H. , Hayashi , T. , Terakami , S. , Takada , N. , Saito , T. , and Yamamoto , T . 2013 , Genomic prediction of trait segregation in a progeny population: a case study of Japanese pear (Pyrus pyrifolia) . BMC Genet ., 14 , 81 . OpenUrl CrossRef 6. ↵ Minamikawa , M. F. , Nonaka , K. , Hamada , H. , Shimizu , T. , and Iwata , H . 2022 , Dissecting Breeders’ Sense via Explainable Machine Learning Approach: Application to Fruit Peelability and Hardness in Citrus . Front. Plant Sci ., 13 , 832749 . OpenUrl 7. ↵ Shiratake , K. , and Suzuki , M . 2016 , Omics studies of citrus, grape and rosaceae fruit trees . Breed. Sci ., 66 , 122 – 38 . OpenUrl 8. ↵ Mondin , M. , Oliveira , C. A. de , and Vieira , M. L. C . 2010 , Karyotype characterization of Malpighia emarginata (Malpighiaceae) . Rev. Bras. Frutic ., 32 , 369 – 74 . OpenUrl 9. ↵ Gladman , N. , Goodwin , S. , Chougule , K. , Richard McCombie , W. , and Ware , D . 2023 , Era of gapless plant genomes: innovations in sequencing and mapping technologies revolutionize genomics and breeding . Curr. Opin. Biotechnol ., 79 , 102886 . OpenUrl 10. ↵ Dudchenko , O. , Batra , S. S. , Omer , A. D. , et al. 2017 , De novo assembly of the Aedes aegypti genome using Hi-C yields chromosome-length scaffolds . Science , 356 , 92 – 5 . OpenUrl Abstract / FREE Full Text 11. ↵ Tang , H. , Zhang , X. , Miao , C. , et al. 2015 , ALLMAPS: robust scaffold ordering based on multiple maps . Genome Biol ., 16 , 3 . OpenUrl CrossRef PubMed 12. ↵ Marçais , G. , and Kingsford , C . 2011 , A fast, lock-free approach for efficient parallel counting of occurrences of k-mers . Bioinformatics , 27 , 764 – 70 . OpenUrl CrossRef PubMed Web of Science 13. ↵ Cheng , H. , Concepcion , G. T. , Feng , X. , Zhang , H. , and Li , H . 2021 , Haplotype-resolved de novo assembly using phased assembly graphs with hifiasm . Nat. Methods , 18 , 170 – 5 . OpenUrl CrossRef 14. ↵ Li , H . 2018 , Minimap2: pairwise alignment for nucleotide sequences . Bioinformatics , 34 , 3094 – 100 . OpenUrl CrossRef PubMed 15. ↵ Simão , F. A. , Waterhouse , R. M. , Ioannidis , P. , Kriventseva , E. V. , and Zdobnov , E. M . 2015 , BUSCO: assessing genome assembly and annotation completeness with single-copy orthologs . Bioinformatics , 31 , 3210 – 2 . OpenUrl CrossRef PubMed 16. ↵ Shirasawa , K. , Hirakawa , H. , and Isobe , S . 2016 , Analytical workflow of double-digest restriction site-associated DNA sequencing based on empirical and in silico optimization in tomato . DNA Res ., 23 , 145 – 53 . OpenUrl CrossRef PubMed 17. ↵ Schmieder , R. , and Edwards , R . 2011 , Quality control and preprocessing of metagenomic datasets . Bioinformatics , 27 , 863 – 4 . OpenUrl CrossRef PubMed Web of Science 18. ↵ Langmead , B. , and Salzberg , S. L . 2012 , Fast gapped-read alignment with Bowtie 2 . Nat. Methods , 9 , 357 – 9 . OpenUrl CrossRef PubMed Web of Science 19. ↵ Li , H . 2011 , A statistical framework for SNP calling, mutation discovery, association mapping and population genetical parameter estimation from sequencing data . Bioinformatics , 27 , 2987 – 93 . OpenUrl CrossRef PubMed Web of Science 20. ↵ Danecek , P. , Auton , A. , Abecasis , G. , et al. 2011 , The variant call format and VCFtools . Bioinformatics , 27 , 2156 – 8 . OpenUrl CrossRef PubMed Web of Science 21. ↵ Rastas , P . 2017 , Lep-MAP3: robust linkage mapping even for low-coverage whole genome sequencing data . Bioinformatics , 33 , 3726 – 32 . OpenUrl CrossRef 22. ↵ Gabriel , L. , Brůna , T. , Hoff , K. J. , et al. 2024 , BRAKER3: Fully automated genome annotation using RNA-seq and protein evidence with GeneMark-ETP, AUGUSTUS, and TSEBRA . Genome Res ., 34 , 769 – 77 . OpenUrl Abstract / FREE Full Text 23. ↵ Stiehler , F. , Steinborn , M. , Scholz , S. , Dey , D. , Weber , A. P. M. , and Denton , A. K . 2021 , Helixer: cross-species gene annotation of large eukaryotic genomes using deep learning . Bioinformatics , 36 , 5291 – 8 . OpenUrl CrossRef 24. ↵ Cantalapiedra , C. P. , Hernández-Plaza , A. , Letunic , I. , Bork , P. , and Huerta-Cepas , J . 2021 , eggNOG-mapper v2: Functional Annotation, Orthology Assignments, and Domain Prediction at the Metagenomic Scale . Mol. Biol. Evol ., 38 , 5825 – 9 . OpenUrl CrossRef 25. ↵ UniProt Consortium . 2023 , UniProt: the Universal Protein Knowledgebase in 2023 . Nucleic Acids Res ., 51 , D523 – 31 . OpenUrl CrossRef 26. ↵ Cheng , C.-Y. , Krishnakumar , V. , Chan , A. P. , Thibaud-Nissen , F. , Schobel , S. , and Town , C. D . 2017 , Araport11: a complete reannotation of the Arabidopsis thaliana reference genome . Plant J ., 89 , 789 – 804 . OpenUrl CrossRef PubMed 27. ↵ Buchfink , B. , Reuter , K. , and Drost , H.-G . 2021 , Sensitive protein alignments at tree-of-life scale using DIAMOND . Nat. Methods , 18 , 366 – 8 . OpenUrl CrossRef PubMed 28. ↵ Altschul , S. F. , Madden , T. L. , Schäffer , A. A. , et al. 1997 , Gapped BLAST and PSI-BLAST: a new generation of protein database search programs . Nucleic Acids Res ., 25 , 3389 – 402 . OpenUrl CrossRef PubMed Web of Science 29. ↵ Cheng , H. , Song , X. , Hu , Y. , et al. 2023 , Chromosome-level wild Hevea brasiliensis genome provides new tools for genomic-assisted breeding and valuable loci to elevate rubber yield . Plant Biotechnol. J ., 21 , 1058 – 72 . OpenUrl 30. ↵ Alves-Pereira , A. , Zucchi , M. I. , Clement , C. R. , et al. 2022 , Selective signatures and high genome-wide diversity in traditional Brazilian manioc (Manihot esculenta Crantz) varieties . Sci. Rep ., 12 , 1268 . OpenUrl 31. ↵ Lu , J. , Pan , C. , Fan , W. , et al. 2022 , A Chromosome-level Genome Assembly of Wild Castor Provides New Insights into its Adaptive Evolution in Tropical Desert . Genomics Proteomics Bioinformatics , 20 , 42 – 59 . OpenUrl CrossRef 32. ↵ Cabanettes , F. , and Klopp , C . 2018 , D-GENIES: dot plot large genomes in an interactive, efficient and simple way . PeerJ , 6 , e4958 . OpenUrl CrossRef PubMed 33. ↵ Alexander , D. H. , Novembre , J. , and Lange , K . 2009 , Fast model-based estimation of ancestry in unrelated individuals . Genome Res ., 19 , 1655 – 64 . OpenUrl Abstract / FREE Full Text 34. ↵ Lee , T.-H. , Guo , H. , Wang , X. , Kim , C. , and Paterson , A. H . 2014 , SNPhylo: a pipeline to construct a phylogenetic tree from huge SNP data . BMC Genomics , 15 , 162 . OpenUrl CrossRef PubMed 35. ↵ Letunic , I. , and Bork , P . 2021 , Interactive Tree Of Life (iTOL) v5: an online tool for phylogenetic tree display and annotation . Nucleic Acids Res ., 49 , W293 – 6 . OpenUrl CrossRef PubMed 36. ↵ Ito , A. , Kajiwara , Y. , Kanmera , S. , et al. 2014 , Identification of Acerola ( Malpighia glabra L.) Accessions by SRAP Markers . Tropical Agriculture and Development , 58 , 30 – 2 . OpenUrl 37. ↵ Ichihara , H. , Yamada , M. , Kohara , M. , et al. 2023 , Plant GARDEN: a portal website for cross-searching between different types of genomic and genetic resources in a wide variety of plant species . BMC Plant Biol ., 23 , 391 . OpenUrl Back to top Previous Next Posted July 30, 2024. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Chromosome-scale genome assembly of acerola (Malpighia emarginata DC.) Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Chromosome-scale genome assembly of acerola ( Malpighia emarginata DC.) Kenta Shirasawa , Kazuhiko Harada , Noriaki Haramoto , Hitoshi Aoki , Shota Kammera , Masashi Yamamoto , Yu Nishizawa bioRxiv 2024.07.30.605750; doi: https://doi.org/10.1101/2024.07.30.605750 Share This Article: Copy Citation Tools Chromosome-scale genome assembly of acerola ( Malpighia emarginata DC.) Kenta Shirasawa , Kazuhiko Harada , Noriaki Haramoto , Hitoshi Aoki , Shota Kammera , Masashi Yamamoto , Yu Nishizawa bioRxiv 2024.07.30.605750; doi: https://doi.org/10.1101/2024.07.30.605750 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genomics Subject Areas All Articles Animal Behavior and Cognition (7843) Biochemistry (18349) Bioengineering (14521) Bioinformatics (43387) Biophysics (22110) Cancer Biology (19231) Cell Biology (26343) Clinical Trials (138) Developmental Biology (13730) Ecology (20542) Epidemiology (2067) Evolutionary Biology (25008) Genetics (15928) Genomics (23128) Immunology (18322) Microbiology (41623) Molecular Biology (17656) Neuroscience (91504) Paleontology (683) Pathology (2928) Pharmacology and Toxicology (4983) Physiology (7933) Plant Biology (15622) Scientific Communication and Education (2077) Synthetic Biology (4457) Systems Biology (10061) Zoology (2336) window.__CF$cv$params={r:'a251bb878cfcd89d',t:'MTc4NTcyMjQ2Ng==',u:'019fc55a795e7d61b4e732b6a266724d',ut:'xETpESUugQWNEtBAx59QvDLJdJtj4exWsbVDKuq4Lx4-1785722468-1.2.1.1-zrGnYKyuJO9ijx_0is8TyZoBtVgnNopD6jSvPWIUHEqxDDV5Lx0ZA.E6BV319pbXt.vq15Dt0dF4JH4nmUWJvh1vlvPK4D7x_0Fy8kAX86U',i:60};(function(){if(!document.body)return;var s=document.createElement('script');s.src='/cdn-cgi/challenge-platform/scripts/precursor/main.js';document.head.appendChild(s);})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.