High-resolution genome and genetic map of tetraploid Allium porrum expose pericentromeric recombination

preprint OA: closed CC-BY-NC-ND-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

We present the first reference genome of highly heterozygous autotetraploid Allium porrum (leek). Combining long-read sequencing with SNP-array screening of two experimental F1 populations, we generated a genetic map with 11,429 SNP markers across 8 linkage groups and a chromosome-scale assembly of Allium porrum (leek) totaling 15.2 Gbp in size. High quality of the reference genome is substantiated by 97.2% BUSCO completeness and a mapping rate of 96% for full-length transcripts. The linkage map exposes the recombination landscape of leek and confirms that crossovers are predominantly proximal located to the centromeres, contrasting with distal recombination landscapes observed in other Allium species. Comparative genomics revealing structural rearrangements between A. porrum and its relatives ( A. fistulosum , A. sativum , A. cepa ), suggests a closer genomic relationship to A. sativum . Our annotated high-quality reference genome delivers crucial insights into the leek genome structure, recombination landscape, and evolutionary relationships within the Allium genus, offering significant implications for breeding programs, facilitating marker-assisted selection and genetic improvement in leek.
Full text 70,618 characters · extracted from preprint-html · click to expand
High-resolution genome and genetic map of tetraploid Allium porrum expose pericentromeric recombination | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results High-resolution genome and genetic map of tetraploid Allium porrum expose pericentromeric recombination View ORCID Profile Ronald Nieuwenhuis , View ORCID Profile Roeland Voorrips , View ORCID Profile Danny Esselink , Thamara Hesselink , View ORCID Profile Elio Schijlen , View ORCID Profile Paul Arens , Jan Cordewener , View ORCID Profile Olga Scholten , View ORCID Profile Sander Peters doi: https://doi.org/10.1101/2025.04.21.649809 Ronald Nieuwenhuis 1 Business Unit of Bioscience, cluster Applied Bioinformatics, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ronald Nieuwenhuis Roeland Voorrips 2 Business Unit of Plant Breeding, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Roeland Voorrips Danny Esselink 2 Business Unit of Plant Breeding, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Danny Esselink Thamara Hesselink 1 Business Unit of Bioscience, cluster Applied Bioinformatics, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site Elio Schijlen 1 Business Unit of Bioscience, cluster Applied Bioinformatics, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Elio Schijlen Paul Arens 2 Business Unit of Plant Breeding, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Paul Arens Jan Cordewener 1 Business Unit of Bioscience, cluster Applied Bioinformatics, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site Olga Scholten 2 Business Unit of Plant Breeding, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Olga Scholten Sander Peters 1 Business Unit of Bioscience, cluster Applied Bioinformatics, Wageningen University and Research , Droevendaalsesteeg 1, 6708 PB Wageningen, The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sander Peters For correspondence: sander.peters{at}wur.nl Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract We present the first reference genome of highly heterozygous autotetraploid Allium porrum (leek). Combining long-read sequencing with SNP-array screening of two experimental F1 populations, we generated a genetic map with 11,429 SNP markers across 8 linkage groups and a chromosome-scale assembly of Allium porrum (leek) totaling 15.2 Gbp in size. High quality of the reference genome is substantiated by 97.2% BUSCO completeness and a mapping rate of 96% for full-length transcripts. The linkage map exposes the recombination landscape of leek and confirms that crossovers are predominantly proximal located to the centromeres, contrasting with distal recombination landscapes observed in other Allium species. Comparative genomics revealing structural rearrangements between A. porrum and its relatives ( A. fistulosum , A. sativum , A. cepa ), suggests a closer genomic relationship to A. sativum . Our annotated high-quality reference genome delivers crucial insights into the leek genome structure, recombination landscape, and evolutionary relationships within the Allium genus, offering significant implications for breeding programs, facilitating marker-assisted selection and genetic improvement in leek. Introduction Leek ( Allium ampeloprasum var. porrum syn. Allium porrum ) is a member of the Allium genus to which also the well-known onion ( A. cepa ), shallot ( A. cepa var. aggregatum ), garlic ( A. sativum ), Japanese bunching onion or Welsh onion ( A. fistulosum ), scallion ( A. cepa var. cepa ), chives ( A. schoenoprasum ) and Chinese onion ( A. chinense ) vegetables belong. Historically, Allium species have been cultivated and appreciated for consumption already from the second millennium BC onward, as evidenced by dried specimens found from archeological sites in Egypt and Mesopotamia ( Hehn, 1870 ; Zohary et al ., 2012 ) and documented Roman recipes ( Sanderson & Renfrew, 2005 ). Today, cultivated leek is globally appreciated for its mild onion-like taste and nutritional value with high content of vitamin B6, C and K, folate, iron and manganese. Leek is a cross-fertilizing species and is generally considered a tetraploid (2n=4x=32) ( Kik et al ., 2021 ). Both auto- and (weak) segmental allopolyploidy are reported in different studies based on varying levels of multivalent formation ( Levan, 1940 ; Kadry & Kamel, 1955 ; Murín, 1964 ; Koul & Gohil, 1970 ). Although leek can self-fertilize to a limited extent of 20%, it displays strong inbreeding depression with plant-weight reductions of up to 60% ( Berninger & Buret, 1967 ). In general, modern leek breeding concentrates on better crop uniformity, higher yield and pest and disease resistance. To obtain better uniformity and higher yield, mostly F1-hydrids are exploited as commercial varieties. Because of intensive selection by leek growers since the last century, a significant amount of genetic variation originally present in the old landraces has been lost in leek germplasm. The genetic erosion has further challenged leek breeders for the identification and introgression of resistance to leek moth (Acrolepiopsis assectella), leek fly (Delia antiqua), thrips (Thrips tabaci ), onion leaf miner ( Liriomyza cepae) stem and bulb nematodes ( Ditylenchus dipsaci, Meloidogyne spp., Pratylenchus penetrans, Paratrichodorus spp., Trichodorus spp. and Longidorus spp.), smudge ( Colletotrichum circinans), leaf blotch (Cladosporium allii), white tip disease (Phytophthora porri), purple blotch (Alternaria porri), leek rust (Puccinia allii), basal rot (Fusarium culmorum), white rot (Sclerotium cepivorum), black mould (Pleospora herbarum), bacterial diseases caused by Pseudomonas spp. and Erwinia spp., and aphid transmitted viral diseases such as Leek Yellow Stripe Virus ( Becue, 2003 ; Lorbeer et al ., 2002 ; Mark et al ., 2002 ; Salomon, 2002 ). Until today resistance to many of these diseases and pests in cultivated leek crops is absent. For genetic improvement of Allium crops, crop wild relatives (CWR) are considered important genetic resources ( Kik et al ., 2021 ). Despite successful use of introgression breeding with cross-fertilizing wild species to transfer beneficial alleles in many target crops, Allium breeders have faced significant challenges in using wide hybridization for leek crop improvement. The success of introgression breeding depends on the biological process known as meiotic recombination, a biological process by which genetic material from the donor into the recipient crop is introduced. The exchange of genetic material between maternal and paternal chromosomes is commonly referred to as crossover (CO). As observed for many organisms, also in Allium species COs are not randomly distributed, and their distribution and localization appear genetically and strictly controlled ( Kudryavtseva et al ., 2023 ). Furthermore, the designation of COs varies considerably between Allium species. Recently, Kudryavtseva and coworkers (2023) demonstrated that A. fistulosum COs predominantly occur in the proximal chromosome regions, while in A. cepa, COs mainly manifest in the distal and interstitial chromosome regions. These differences may result in homoeologous chromosome pairing disorders during meiosis in hybrids, leading to reduced COs in regions containing beneficial alleles thus hampering introgression breeding. Indeed, Allium introgression lines often display genome instability through a phenomenon known as genome dominance, manifested by unequal inheritance and loss of introgressed genetic donor material in successive generations, further impeding the employment of wide hybridization in practical Allium breeding ( Kopecký et al ., 2022 ). Thus, understanding of the recombination and CO landscapes is crucial for successful introgression breeding in Allium crops. The development of a reference genome and genetic map offers a significant opportunity to overcome many of these challenges in leek breeding. These tools enable the assessment of genetic variation and inheritance, facilitating more efficient breeding strategies. A major advantage is the improved ability to identify beneficial haplotypes for disease resistance. Given leek’s widespread susceptibility to pests, fungi, bacteria, and viruses, distinguishing between parental haplotypes in an outcrossing autotetraploid would allow breeders to track, select, and monitor the retention of specific resistance alleles over generations. Additionally, linkage maps help clarify recombination patterns and CO distribution. Since COs are not randomly distributed, a genome assembly and genetic map provide insight into CO patterns, allowing breeders to select compatible breeding lines for successful introgression breeding. This is particularly helpful when incorporating alleles from CWR and avoiding homoeologous chromosome pairing issues ( Jones et al ., 1996 ; Khazanehdari & Jones, 1997 ; Kiełkowska, 2012 ). Linkage maps also help identify and track deleterious recessive alleles that may negatively affect plant fitness, aiding breeders in selecting complementary parental lines with optimized agronomic performance and disease resistance, while minimizing inbreeding depression. Furthermore, structural variation detection is another critical advantage of reference genomes and genetic maps. Large-scale genomic variations, such as copy number variations, inversions, and translocations, can influence key agronomic traits. By combining detailed genome sequences with genetic map data, breeders can better understand these variations and selectively retain or eliminate structural variants that impact crop performance. Genetic maps also play a crucial role in stabilizing introgression lines, allowing the effective transfer of beneficial traits by tracking wild allele inheritance and identifying regions prone to instability. The sheer size, autotetraploid heterozygous nature, and highly repetitive structure of the leek genome have long posed challenges for genome assembly and genetic map construction. In this paper, we present the construction of the first chromosome-scale reference genome for leek, along with the development of a genetic map, marking a significant milestone in Allium genomics and plant breeding. By resolving the complexities of this genome at a chromosome scale, this study provides a foundation for dissecting genetic variation, understanding recombination landscapes, and accelerating breeding efforts. The ability to phase haplotypes and map crossovers will allow for more precise tracking of beneficial alleles, especially in the context of introgression breeding, where homoeologous recombination, chromosome synapsis issues, and genome instability have previously hindered progress ( Jones et al ., 1996 ; Khazanehdari & Jones, 1997 ; Kiełkowska, 2012 ; Kopecký et al ., 2022 ). Moreover, the high level of genomic resolution enables the identification of structural variants that influence agronomic traits, facilitating targeted selection strategies. The reference genome and genetic map presented here are vital resources for leek improvement, set a precedent for genomic studies in complex Allium species, and complement ongoing efforts in onion, garlic, and Welsh onion ( Sun et al ., 2020 ; Hao et al ., 2023 ). Ultimately, this achievement paves the way for more resilient, genetically diverse, and high-performing leek cultivars, ensuring the sustainable advancement of this globally important crop. Materials and methods Plant material We established a genome sequence from an individual leek plant of accession “Leidse prei 2018-94”, a landrace/grower selection from the Netherlands. Linkage mapping was performed in two populations: one F1 of the cross 34012-7 (a genic male sterile (gms) plant donated by Bejo Zaden, the Netherlands) x the “Leidse prei 2018-94” plant mentioned above, consisting of 187 plants, and one F1 of the cross 17169-29 (a gms plant donated by Nunhems / BASF crop science, the Netherlands) x A. ampeloprasum , consisting of 248 plants. DNA isolation & library prep Seven clones of the leek father plant “Leidse prei 2018-94" were used for tissue harvest. Multiple young most inner leaves were harvested, snap frozen in liquid nitrogen and stored at -80°C. Approximately 1.5 g of leaf material was ground with liquid nitrogen and used for HMW gDNA isolation using NucleoBond HMW DNA kit following manufacturer’s instructions (Machery-Nagel). Additionally, 1 g of leaf material was ground with liquid nitrogen and used for HMW gDNA isolation using MagMAX Plant DNA kit following manufacturer’s instructions (Applied Biosystems). DNA quantity, purity and fragment size was determined using Qubit (Invitrogen), OD values (NanoDrop) and Fragment Analyzer (Agilent). Obtained gDNA was sheared in different rounds by Megaruptor 2 (Diagenode) with 20 to 30 kbp target fragment size. Small fragments were removed by utilizing Blue Pippin (Sage Science) for size selection. Size selected DNA was used for six SMRTbell library preps using SMRTbell Express Template Preparation kit 2.0 according to manufacturer’s instructions (Pacbio; Procedure-Checklist-Preparing-HiFi-SMRTbell-Libraries-using-SMRTbell-Express-Template-Prep). SMRTbell libraries were subjected to DNA polymerase complexing using Sequel binding kit v2.0/3.0 and Sequel Polymerase 3.0/2.2. Final sequencing was done using sequence primer v4 /v5 on a Pacbio Sequel-II/ Sequel-IIe instrument (see section Genome sequencing). Sequencing reactions were performed with sequencing kit 2.0 /3.0, using adaptive loading, 65 to 85 pM on plate loading concentration, 120- or 240-minutes immobilization time and 30 hours movie time per SMRT cell. Genome sequencing Long insert libraries of 15 and 19 kbp were sequenced on the PacBio Sequel-II(e) platform using 44 SMRT cells (Supplementary Table S1). Subsequent circular consensus sequences (CCS) were called using the pbccs v5.0.0 command line utility. HiFi reads were defined as CCS reads having a minimum number of 3 passes and a mean read quality score of Q20. Quality control per SMRTcell was checked via SMRTlink and an in-house pipeline including FastQC v0.11.9 ( Andrews, 2010 ), KMC v3.1.1 ( Kokot et al ., 2017 ), Smudgeplot v0.2.3dev_rn ( Ranallo-Benavidez et al ., 2020 ), GenomeScope v2 ( Ranallo-Benavidez et al ., 2020 ) and BlastN v2.11.0+ ( Altschul et al ., 1990 ; Camacho et al ., 2009) with the NCBI nt, plastid and mitochondrion publicly available databases downloaded 2021-04-02 ( Sayers et al ., 2024 ). Reads produced from different libraries were then combined into a single data set for further analyses. Contig assembly Combined reads of all sequenced libraries were assembled using hifiasm v0.15.1-r334 ( Cheng et al ., 2021 ) with output settings for primary and alternative assembly selected. All contigs from both the primary and alternative assemblies were screened for contamination using NCBI Foreign Contamination Screen v0.4.0 ( Astashyn et al ., 2024 ), and non-subject sequences were subsequently removed. Splitting of contigs was done with NCBI Adaptor v0.4.0 in case of detected adaptor sequences. Any remaining adaptors sequences were removed to avoid assembly errors. BUSCO v5.2.2 ( Manni et al ., 2021a , 2021b ; Simão et al ., 2015 ) was used with the embryophyte odb10 database (2020-09-10) to check completeness of conserved single copy orthologs. Further purging of the primary assembly was performed in a customized manner to accommodate the large dataset and limited sequence coverage. The method optimized single copy BUSCO gene percentage in a minimal set of contigs. (Supplementary Figure S1). Selected contigs were used for subsequent annotation and scaffolding. K-mer based completeness checks were omitted due to the heterozygous and autopolyploid nature of the sequenced specimen, with the purged result representing only a pseudo-haploid version of the genome. Reference based assembly was applied to the leek WGS HiFi dataset to assemble the mitochondrion and chloroplast genomes. For the plastid genome ptGAUL v1.0.5 ( Zhou et al ., 2023 ) was used with the A. ampeloprasum reference sequence (NC_044666.1). For the mitochondrial genome mitoHIFI v3.0.1 ( Uliano-Silva et al ., 2023 ) and ptGAUL v1.0.5 assemblers were used with reference from A. cepa and A. sativa . RNA isolation & library prep For RNA isolation, material of different tissues (green leaf, stem section green / white, basal plate section, roots) from leek father plant (2018-94-02) were collected. In addition, flower tissue was collected from a different plant (leek mother plant 1716929 ms), immediately snap frozen in liquid nitrogen, and stored at -80°C. RNA from the father plant tissues was isolated using Ambion PureLinkRNA Mini Kit (Life Technologies). RNA from the mother plant flower tissue was isolated using ZymoBIOMICS RNA Mini prep Kit (Zymo Research). Subsequently, RNA quantity and purity was analyzed by Qubit (Invitrogen), OD values (NanoDrop) and Bioanalyzer RNA plant pico assay (Agilent). Of each tissue, 300 ng total RNA was used to create a barcoded IsoSeq library following manufacturer’s guidelines (Pacbio; Procedure-Checklist-Iso-Seq-Express-Template-Preparation-for-Sequel-and-Sequel-II-Systems). SMRTbell library yield was quantified by Qubit (Invitrogen) and SMRTbell sizes were checked by Bioanalyzer High sensitivity DNA Assay (Agilent). Libraries were pooled equimolar, subjected to DNA Polymerase SMRTbell complexing using Sequel II Binding kit 2.0. and primer v4 prior to loading on 4 SMRT cells with 50 to 58 pM on plate loading concentration. Sequencing reaction was performed on a PacBio Sequel-II system with 24-hour movie time. Transcriptome sequencing Genetic diversity was mined for array probe design by using RNA-seq of 14 A. porrum accessions including the mapping population parents “Leidse prei 2018-94”, MS34012 -7, and species A. lusitanicum and A. eduardii (Supplementary Table S2). RNA-seq libraries of young plantlets and leaf tissue were sequenced on an Illumina Novaseq6000 platform using an S2 flow cell. Base calling and initial quality filtering of raw sequencing data was done with bcl2fastq v2.20.0.422 ( https://emea.support.illumina.com/downloads/bcl2fastq-conversion-software-v2-20.html ) using default settings. Next, full-length transcripts from different tissues of the sequenced subject were sequenced, using PacBio IsoSeq on a Sequel-IIe platform for annotation of the de novo assembled genome. Consensus reads were called with the ccs v5.0.0 command-line utility of PacBio. HiFi reads were generated using the same specifications as used for the genomic reads. Reads were then stripped of primers and demultiplexed using lima v2.0.0 ( https://github.com/PacificBiosciences/barcoding ). Poly-A tails were trimmed and concatemers were removed using isoseq3 v3.4.0 ( https://github.com/PacificBiosciences/IsoSeq ) to generate full length non-concatemer reads, which were subsequently clustered using isoseq3 cluster ( https://github.com/ylipacbio/IsoSeq3 ) without final polishing. Axiom array design To create a genetic map, first a reference transcriptome sequence set was constructed by clustering all the IsoSeq data of “Leidse prei 2018-94” with isONclust ( Sahlin & Medvedev, 2020 ). This reference formed the basis for variant calling. A set of largest 45K clusters was then selected based on the completeness and duplication rate of the BUSCO score (74% completeness with 10% duplication rate). The final reference sequence set, consisting of the largest 45K isONclust sequences, was concatenated with the chloroplast genome of A. ampeloprasum to accommodate subsequent read mapping and filtering ( Filyushin et al ., 2019 ). IsoSeq reads of samples 2018-94-02 and MS17169-29-2 were mapped using Minimap2 v2.11-r797 (Li, 2018) with additional settings (splice, -secondary=no, -C5 -06,24, -B4). Illumina reads of samples MS 17169-29-2, MS 34012-7 and 2018-94-1-1 were mapped using STARv020201 ( Dobin et al ., 2013 ) with default settings. Duplicate read pairs were marked with Picard tools (v2.2.1) MarkDuplicates ( https://broadinstitute.github.io/picard/ ). The bam files were screened with samtools ( Danecek et al ., 2021 ) to filter for non-primary and supplementary alignments (sam flag -F2304). NGSEP3 v4.01 ( Tello et al ., 2019 ) MultisampleVariantsDetector was used for variant calling of the parents of the two segregating populations (default settings with -ploidy=4). Biallelic SNP variants were selected having a flanking sequence of at least 30 bases without additional SNPs or indels. Probes with a low complexity sequence, low GC-content, having an A/T or C/G variant, or showing redundancy, were subsequently discarded. For the design of the SNP array, three priorities were defined: Priority 1 focused on probes that needed to be called in both sets, Priority 2 prioritized probes called in the parents of the first cross, and Priority 3 prioritized probes in the parent of the second cross. Following the probe design, a draft de-novo assembly of a non-purged genome became available that was used to filter the recommended probes. For that the mapped probes were screened with Bowtie2 (--very sensitive) ( Langmead & Salzberg, 2012 ) over possible intron/exon boundaries. Probes with more than the expected maximum of 4 possible hits in a tetraploid individual were filtered out. maximum number of 350 probes per contig were chosen. Dosage calling The array hybridization was performed in two separate experiments, one with each of the two F1 populations, along with replicate samples of the parents and in the case of the second F1 population also with unrelated material of different leek types. Dosage calling was performed with the R package fitPoly ( Voorrips et al ., 2011 ; Zych et al ., 2019 ) Linkage mapping Linkage mapping in both F1 populations was performed using the R package polymapR ( Bourke et al ., 2018 ), according to the vignette of that package. The procedure consists of the following consecutive steps: filtering against markers and SNPs with too many missing data; merging duplicate F1 individuals; binning duplicate markers; assigning simplex x nulliplex (and analogous) markers to chromosomes and then to homologues for the two parents separately; matching the maternal and paternal chromosomes using simplex x simplex markers; assigning all other marker types to the simplex-nulliplex homologues; ordering the markers on the chromosomes; fine-tuning by successively eliminating ill-fitting markers. Finally, the duplicated markers that were set aside in the binning step were added back at the position of the marker representing the bin. A consensus linkage map over the two populations was created by combining the pairwise linkage data (i.e., for each marker pair the recombination fraction and the LOD score) of the corresponding chromosomes from both populations. For marker pairs that occurred in the linkage data of both populations the combined recombination and LOD score were calculated by averaging the separate values, weighted by the squares of the LOD scores. With the combined linkage data marker ordering and fine-tuning was performed again. Diagnostic plots were produced and a test for preferential pairing was performed, using functions from the polymapR package. Linkage maps were plotted using MapChart ( Voorrips, 2002 ). Scaffolding Selected contigs from the purging step were used as reference for mapping of the probe sequences of Axiom markers present on the final map. The probes were mapped using GMAP v2021-08-25 ( Wu & Watanabe, 2005 ) with a large index, suitable for a genome larger than 232 Mb. They were then filtered to ensure a coverage of at least 0.98, an identity of at least 0.95, and only one mapping position. The mapping was combined with the original map and subsequently scaffolded using ALLMAPS v4 ( Tang et al ., 2015 ). Public reference genomes of A. sativum , A. fistulosum and A. cepa were obtained from NCBI RefSeq entries GCA_014155895.2, GCA_030737875.1, GCA_030737815.1. and GCA_030765085.1. Mapping of A. porrum markers against those assemblies was done similarly to the mapping described for scaffolding. Annotation Repeats in the genome assembly were annotated using the RepeatModeler, RepeatClassifier and RepeatMasker tools using the combined RepBase (2014) and Dfam (2020) databases for classification of identified repeats ( Bao et al ., 2015 ; Hubley et al ., 2016 ; Smith et al ., 2013 ). Full length non-concatemer IsoSeq reads were mapped against the genome assembly using minimap2 -ax splice -uf -- secondary=no -C5 . Braker v3 pipeline was then used for building gene models, gene prediction and gene annotation (Gabriel et al ., 2024; Kovaka et al ., 2019 ; Pertea & Pertea, 2020 ; Quinlan, 2014 ; Stanke et al ., 2006 , 2008 ). Efforts to reveal the centromeres, (sub-)telomeres and ribosomal DNA arrays were based on BLASTn searches using NCBI accessions MT374061.1 , MT374062.1, MH017541.1 and MH017541.1 ( Fu et al ., 2019 ; Kirov et al ., 2020 ). Ribosomal DNA arrays were annotated using Infernal v1.1.5 ( Nawrocki & Eddy, 2013 ) and selected eukaryote 5S, 5.8S, 18S and 28S sequences from the RFAM database. Results De novo assembly Sequencing of “Leidse prei 2018-94” DNA yielded 883 Gbp of PacBio HiFi data divided over 57.5 million reads (Supplementary Table S1, Supplementary Figure S2 & S3). In total, 4.2% and 5.0% of the hits were to plastid and mitochondrial databases respectively. Quality control showed an average read quality of Q30, which is generally considered a sufficient accuracy level for faithful genome reconstruction. K-mer analysis for k-mer size 21 revealed a GenomeScope profile showing the highest peak at a coverage of 14 and two additional minor peaks at 28 and 42 (Supplementary Figure S4), representing genomic sequences present in single copy or two and three copies, respectively. Modelling of the genome parameters such as genome size, heterozygosity and repetitiveness using the obtained distribution proved to be unsuccessful as the peaks did not diverge sufficiently, and only one broad peak was captured by the Genomescope2 model. K-mer analysis revealed strong AAAB and AB signals confirming the autotetraploid history of leek (Supplementary Figure S5). Rough estimation of genome size based on the total data set size and the first peak, under the assumption of full heterozygosity, was a 4n complement of 63 Gbp with 1n equaling to 15.75 Gbp. De novo assembly resulted in a total assembly size of 70.7 Gbp consisting of 426,980 contigs with an N50 of 3.7 Mbp ( Table 1 ). Primary and alternative assemblies sized to 38.9 and 31.9 Gbp with N50 sizes of 27.5 and 0.14 Mbp respectively. The largest contigs with an L50 index of 377 in the primary assembly, further supported the assignment of the primary assembly (Supplementary Figure S6). Contamination screening identified 49 and 2 hits to proteobacteria for primary and alternative assemblies respectively, which were removed. Additionally, 543 and 231 PacBio adaptor and primer sequences were detected in the primary and alternative contig sets and contigs were consequently either trimmed or broken. Single copy ortholog benchmarking, referencing the embryophyta lineage set, showed a very high completeness score of 97.2%, but also a high duplication rate of 95.2% as can be expected from a polyploid. For successful scaffolding with a linkage map, marker sequences must map to a single position in the genome and polyploid derived duplication must be minimized. To satisfy this criterion, we applied custom purging based on the BUSCO gene set, selecting 369 contigs (totaling 16.1 Gbp) while maintaining a high benchmark completeness score and reducing the duplication rate to 38.3%. While still a suboptimal percentage, this purging result combined with further selection on uniquely mapping markers would enable us to scaffold a pseudohaploid genome. The mapping rates of IsoSeq full-length transcripts against both the primary and purged contig sets were 96%, indicating that purging did not significantly impact transcriptome representation. This ensured that scaffolding and the final assembly retained high genomic integrity and transcript completeness. View this table: View inline View popup Download powerpoint Table 1: Assembly statistics for all subsequent stages of nuclear genome assembly and of chloroplast assembly Assembly of the organellar genomes was successful only for the chloroplast but not for the mitochondrion. The reference-based chloroplast assembly has a total length of 152,493 bp and follows the conserved pattern of a long single copy section (LSC) and short single copy section (SSC) interspersed by an inverted repeat. This result indicates that the chloroplast genome is highly conserved and structurally resembles known chloroplast reference genomes. This allowed for successful assembly through reference-based methods, whereas the failure to assemble the mitochondrial genome may suggest greater complexity or divergence in its structure, possibly requiring more specialized assembly approaches or higher-quality data. Linkage mapping The linkage maps of both F1 populations, and the integrated map, consisted of 8 linkage groups, corresponding to the eight chromosomes of leek. Overall statistics of the maps are shown in Table 2 . Diagnostic plots for the integrated map are shown in Supplementary Figure S7, and scatterplots of the Population 1 and Population 2 maps versus the integrated map in Supplementary Figure S8. The integrated map and the F1 population 1 map, the latter having the target “Leidse prei 2018-94” plant as father, were quite similar, although the integrated map contains more markers. The length of the chromosomes in both maps did not differ much. The map of F1 Population 2 showed some significant differences with the integrated map and the Population 1 map. The map of Population 2 was much sparser, and chromosomes 6 and 8 in this map were much shorter than the other maps. Also, the lengths of chromosomes 3 and 7 were larger than those of the other two maps, containing larger gaps. These map features point to a lower quality for the map constructed from the second population. The integrated map of chromosome 6 is based only on the linkage data of F1 population 1; it is identical to it, except that a few markers that were added in Population 2 showed an identical segregation as the markers present in the Population 1 map. View this table: View inline View popup Download powerpoint Table 2: Linkage map statistics The markers on the integrated linkage map were mostly concentrated in dense clusters near the ends of all chromosomes ( Figure 1 ). A similar distribution was observed in the separate maps for the first and second F1 populations. However, in the map of the second population, the left distal regions of chromosomes 6 and 8 were missing (not shown). Since the markers were derived from expressed sequences, this suggests that a large part of the expressed genes are in blocks with low recombination frequency. In Population 1 the occurrence of double reduction was studied based on simplex x nulliplex markers of both parents. At both telomeres of all chromosomes a level of 4%-5% double reduction was observed, indicating the occurrence of quadrivalents and polysomic inheritance. Download figure Open in new tab Figure 1: Marker density in cM/locus for the eight linkage groups established in the integrated genetic map of Allium porrum. Each vertical stripe represents one marker. Highly dense regions tend to color black and generally can be seen at both distal ends and center of each linkage group. Scaffolding Mapping of marker sequences to 369 selected contigs, representing a minimal set with maximum BUSCO completeness, showed that 11,143 out of 11,429 markers (97.5%) were successfully mapped. Further filtering based on alignment coverage, identity, and mapping ambiguity removed 835, 300, and 3,123 markers, respectively. The remaining 7,184 mapped markers were used to orientate and order the 369 contigs with the linkage map as template to construct a chromosome-level genome assembly. Of the 316 contigs anchored with the map, 303 could be oriented using at least two aligned markers. The placed contigs represented 94.3% of the total sequence length, with 15.9 Gbp of the 16.2 Gbp scaffolding input mapped to the linkage map. Of this, 15.3 Gbp was anchored to chromosomes, and 15.0 Gbp was also oriented. Unplaced sequences included 25 contigs lacking marker information and 28 contigs were omitted due to unresolved conflicts with the linkage map. The largest chromosome, chromosome 2, measured 2.45 Gbp, while the smallest, chromosome 8, was 1.40 Gbp. Overall, the correlation between genetic and physical positions was high, with Spearman’s rank correlation coefficient (ρ) ranging from 0.96 for chromosome 4 to 0.66 for chromosome 6, with most chromosomes exceeding 0.9 (Supplementary Figure S9). Repeat and gene annotation De novo repeat modeling and annotation, using RepBase and curated DFAM databases, identified 4,312 repeat models. The largest group remained unclassified, comprising 1,725 unknown repeat families and 1,688 unknown LTR retrotransposon families. Among the classified repeats, 391 belonged to the LTR/Copia family, while 361 were classified as LTR/Ty3 (formerly known as Gypsy). Overall, repeat masking covered 81.51% of the leek genome assembly, highlighting its highly repetitive nature, dominated by both known and unknown transposable elements. Gene annotation of the repeat masked genome resulted in 66,021 genes of with an average length of 6,553 bp and a mean 3.7 exons per gene. The repeat and gene distribution can be described as uniformly distributed along the physical chromosomes and no clear evidence was found of centromere and telomere associated sequences (Supplementary Figure S10). Scaffolded genome reveals distinct recombination landscape The Marey maps ( Figure 2 ) show a comparison between the generated linkage map and the physical map that was obtained by de novo assembly. Overall, the maps indicate a sufficient coverage of markers along the chromosomes. The densely populated areas on the linkage map make up the chromosome ends. Because the array design was based on transcriptome derived variants, the marker distribution shows the density of genic regions which are present quite uniformly along the physical chromosomes in leek (with slightly higher densities in distal parts of the linkage map). For all chromosomes a full sigmoid like pattern can be observed, with the exception of chromosome 6. This chromosome shows a partial sigmoid like pattern with a flat part extending the chromosome end. Most of the chromosomes lack markers in the middle section, possibly representing the centromere region. The flatter parts at the chromosome ends indicate a smaller distance between successive markers on the genetic map, while the middle parts of each chromosome show relatively larger genetic distances between successive markers. This pattern thus points to a relatively high recombination frequency in the middle of the chromosome compared to the chromosome ends. Apparently, recombination in A. porrum occurs mainly in the regions directly adjacent to the centromere and not in the distal chromosome ends, consistent with the cytological observation that chiasmata in A. ampeloprasum are predominantly localized proximally to the centromere ( Kollmann, 1972 ). Download figure Open in new tab Figure 2: Marey maps displaying physical (x-axis) and genetic positions (y-axis) of markers. Density estimates are displayed along the axes for each chromosome. The sigmoid-like patterns reveal the proximal recombination landscape in A. porrum. Density estimates show a large number of markers in regions with minimal recombination on the genetic map. Genome comparison to A. fistulosum , A. sativum and A. cepa A comparison of physical marker positions from our genetic map with physical positions on other recently published high-quality genome assemblies provides insight into large structural chromosome rearrangements. Mapping the marker sequences from the final genetic map to Allium fistulosum (Welsh onion/bunching onion) ( Liao et al ., 2022 ; Hao et al ., 2023 ) yielded 1,135 unambiguous mapping positions. As shown in Figure 3 , large intrachromosomal inversions are present on chromosomes 1, 2, 3, 5, and 6. Although a few markers suggest possible translocations by mapping to different chromosomes—indicated by deviations in the color code displayed in the chromosome synteny panels of Figure 3 —no major interchromosomal rearrangements were observed between leek and bunching onion ( A. fistulosum ). Download figure Open in new tab Figure 3: Mapping of markers on A. porrum (y-axis) and A. fistulosum (x-axis) reveal several large structural variations such as inversions of 1 Gbp on chromosomes 1, 2 and 5. Chromosomes 4 and 7 seem to be overall more colinear at this resolution. Furthermore, a comparison between the final genetic map and A. sativum ( Sun et al ., 2020 ) identified 4,785 mappable marker positions, revealing synteny between leek chromosomes 1, 2, 3, 6, and 8 with A. sativum chromosomes 6, 3, 8, 4, and 1, respectively. However, these chromosomes also exhibit synteny breaks due to intrachromosomal inversions (Supplementary Figure S11). Additionally, large interchromosomal rearrangements were observed for leek chromosomes 4, 5, and 6. Specifically, leek chromosome 2 was partially syntenic with A. sativum chromosomes 2 and 7, leek chromosome 5 showed partial synteny with A. sativum chromosomes 2 and 5, and leek chromosome 6 was partially syntenic with A. sativum chromosomes 5 and 7. However, a comparison with A. sativum from Hao and coworkers (2022) identified 5,013 shared markers but did not reveal intrachromosomal rearrangements (Supplementary Figure SX). This discrepancy may reflect differences in genome assembly quality, as suggested by Hao et al . (2022). Alternatively, it is possible that these differences result from accession-specific rearrangements in A. sativum . A comparison with A. cepa (Hao et al ., 2022) identified 1,116 shared mappable markers and revealed synteny breaks due to intrachromosomal rearrangements across all chromosomes (Supplementary Figure S11). In contrast, no major interchromosomal rearrangements were observed between leek and onion. Considering the higher number of shared markers and the smaller size of inversions, A. ampeloprasum appears to be more closely related to A. sativum than to A. cepa or A. fistulosum . This result is consistent with the phylogenetic relationships among Allium species as previously reported by Hirschegger et al . (2010) . Discussion First chromosome-scale reference genome of A. porrum The first genome sequence of a highly heterozygous tetraploid Allium porrum is a valuable addition to the collection of Allium genomes that have currently been reconstructed, as the current collection does not contain such a complex genome yet. The total size of the 15.26 Gbp anchored genome is slightly smaller than the recently published genomes of onion (15.78 Gbp) and garlic (15.52 Gbp), but bigger than the Welsh onion genome of 10.48 Gbp. Compared to the N50 contig size of 81.66, 109.82 and 507.27 Mbp for onion, garlic and Welsh onion respectively, the N50 leek contig size of 57.37 Mbp is notably smaller. This is partly due to the relatively low coverage per haplotype combined with the high heterozygosity and possibly a smaller sequence library insert size. Since the Allium genomes reported by Hao et al., 2023 were also constructed by HiFi based technology, we regard the difference in N50 contig size less likely to be the result of the sequencing technology used. Several findings have reported on genome size for A. ampeloprasum . Our leek genome assembly size significantly exceeds the flow cytometric values reported by Arumuganathan & Earle (1991) , who found a 2C value of 50.27 pg for A. ampeloprasum , corresponding to a single-copy genome size of 12.29 Gbp. Ricroch et al . (2005) reported that the genome size of tetraploid A. porrum is 50.7 ± 0.7 pg, corresponding to n =12.45 Gbp, and additionally noted that the diploid A. ampeloprasum has a haploid genome size of 16.37 Gbp. Ohri et al . (1996) summarized findings on tetraploid A. porrum , which range from 11.78 Gbp to 31.93 Gbp for a single genome copy. Overall, the size of the leek genome assembly we report here falls within the previously reported genome size ranges for both A. ampeloprasum and A. porrum . The use of PacBio HiFi reads with an average read quality score of Q30 was a key factor contributing to the high quality of the reconstructed leek genome. The high quality of the presented assembly is substantiated by the high completeness of conserved single copy ortholog (BUSCO) genes. The 97.2% completeness slightly exceeds the benchmark results for onion (96.4%), garlic (92.6%) and Welsh onion (96.6%) genomes. Our annotation efforts show that the number of 66k genes in leek is in the same range as found by Hao in garlic and onion, which are similar sized genomes. Annotation of repetitive sequences is different however, as the percentage of the genome marked as repetitive is around 11-13 percentage points lower. Marker-based scaffolding with probe sequences designed from transcriptome data and BUSCO-based purging are possible causes for the exclusion in the scaffolding process of contigs consisting of mainly repeat sequence. This depletion would lower the total repeat sequence relatively to the total assembly size. In addition to the high BUSCO scores, the high collinearity with an integrated map from two mapping populations confirms that the order of markers on the generated contigs aligns precisely with linkage calculations. Furthermore, the successful anchoring and orientation of a significant proportion (94.3%) of the genome assembly onto the integrated linkage map, along with the high correlation between genetic and physical positions, demonstrates the robustness and accuracy of the assembly process. The few unresolved contigs and lower correlation on chromosome 6 suggest areas for further refinement, but overall, the results provide a highly contiguous and well-ordered chromosome-level assembly that will support downstream genomic analyses. Marker-rich genic regions in distal clusters on linkage maps and recombination predominantly occurring near centromeres Understanding gene distribution and recombination behavior across a wide range of Allium genomes is crucial for developing effective breeding programs. Furthermore, insight into marker distribution can aid in more efficient breeding by introducing marker-assisted selection. By leveraging dense markers in gene-rich areas, breeders can more accurately track desired traits, improving the speed and efficiency of developing new leek varieties. The markers we developed from mapped transcript sequences primarily target expressed leek genes. Surprisingly, we observed a pronounced clustering of markers towards the chromosome ends on the linkage map, whereas recombination predominantly occurred proximal to the leek centromere ( Figure 2 ). This clustering of markers is partly the result of a relatively low recombination frequency towards the ends of the leek chromosomes, as we observed a relatively more even marker distribution on the physical map ( Figure 2 ). Nonetheless, this contrast suggests that while breeders may focus on gene-rich regions for selecting desirable traits, recombination in the distal leek chromosome regions is limited, potentially constraining reshuffling of traits to new generations. Consequently, understanding the genetic landscape of Allium species is essential for optimizing breeding approaches. Notably, variations in recombination behavior have been observed across different Allium species and subspecies. In tetraploid A. porrum , Levan (1940) reported a predominantly proximal localization of chiasmata, being cytological manifestations of crossovers. Further cytological studies by Kollmann (1972) on A. ampeloprasum subspecies ( ampeloprasum and truncatum ) revealed distinct recombination patterns: while A. ampeloprasum exhibited chiasmata localized near the centromere, A. truncatum displayed subterminal or terminal localization. Interestingly, within each subspecies, chiasma localization appeared independent of ploidy level. In addition, pollen abortion analyses revealed pollen viability to be the most stable in tetraploids and less for other ploidy levels for both subspecies. In addition, the predominant proximal localization of chiasmata seems to reduce multivalent chromosome formation in 4x, 5x, and 6x plants of subsp. ampeloprasum . However, no results pointing to a correlation between chiasma localization and pollen viability were reported ( Kollmann, 1972 ). Profound differences in recombination behavior have also been observed between diploid Allium species. In onion ( A. cepa ), crossovers predominantly occur in the distal chromosome regions, whereas in A. fistulosum , they are mainly localized proximally ( Emsweller and Jones, 1935a ; Kudryavtseva et al ., 2023 ). Moreover, chiasmata in an A. cepa × A. fistulosum hybrid were found to shift significantly toward the distal regions of A. fistulosum homologous chromosomes, suggesting genetic control over crossover localization. This observation supports the hypothesis of Emsweller & Jones (1935b) that A. cepa and A. fistulosum possess dominant and recessive genes, respectively, regulating distal and interstitial chiasmata localization. Their study furthermore showed that progeny from backcrosses of cepa x fistulosum hybrids to cepa and fistulosum exhibited pronounced differences in blooming and fertility. However, no correlation was found between fertility and percentage of good pollen. Interestingly, results suggest a correlation between fertility and chiasma localization as the most fertile plants displayed interstitial chiasmata, while sterile plants displayed terminal chiasmata. However, any conclusion from these observations should be drawn with caution, as the number of plants studied was rather limited ( Emsweller & Jones, 1935b ). Nonetheless, these recombination differences across Allium species may have significant implications for breeding and warrant further investigation into Allium meiotic recombination landscapes. A comparison of our genetic and physical chromosome maps of A. porrum ( Figure 2 ) further supports the previously reported findings and indicate that recombination occurs predominantly in the central parts of leek chromosomes, containing functional centromeres as demonstrated previously by Levan (1940) . Subsequent FISH analysis in A. cepa and A. fistulosum showed the co-localization of repetitive centromeric sequences like those found in many plant species ( Kirov et al ., 2020 ), suggesting that recombination in leek also occurs mostly proximally to the centromeres. We were however unable to obtain any conclusive results on the inclusion of centromeres in our assembled genome, presumed to be caused by the absence of markers in the centromere itself. In this study, we combined information on recombination frequencies between markers obtained through linkage mapping with their physical genome sequence positions, further confirming the proximal localization of recombination in leek chromosomes and further substantiating the existence of a dichotomy in recombination behavior in Allium . To our knowledge, this is the first such comparison in the Allium genus. Supplementary materials Tables: View this table: View inline View popup Download powerpoint Figures: View this table: View inline View popup Download powerpoint Data availability The genomic and transcriptomic data used in this study are available under NCBI BioProject accession PRJNA1231124. Conflict of interest The authors declare no conflict of interest. Acknowledgements This project was supported by the Netherlands Top Consortium for Knowledge and Innovation (TKI project TU-18013). We also thank ENZA Zaden Research & Development B.V., Bejo Zaden B.V., Hazera Seeds B.V., and Nunhems Netherlands B.V. for their support. Funding Topsector Tuinbouw & Uitgangsmaterialen , TU-18013 Abbreviations LOD log of odds SNP single nucleotide polymorphism HiFi High fidelity HMW high molecular weight CO Crossover References ↵ Altschul , S.F. , Gish , W. , Miller , W. , Myers , E.W. & Lipman , D.J . ( 1990 ) “ Basic local alignment search tool .” J. Mol. Biol . 215 : 403 – 410 OpenUrl CrossRef PubMed Web of Science ↵ Andrews , S. , ( 2010 ) FastQC: a quality control tool for high throughput sequence data , http://www.bioinformatics.babraham.ac.uk/projects/fastqc ↵ Arumuganathan , K. & Earle , E.D . ( 1991 ) Estimation of nuclear DNA content of plants by flow cytometry . Plant Molecular Biology Reporter , 9 , 229 – 241 . doi: 10.1007/BF02672073 . OpenUrl CrossRef ↵ Astashyn , A. , Tvedte , E.S. , Sweeney , D. et al. ( 2024 ) Rapid and sensitive detection of genome contamination at scale with FCS-GX . Genome Biol , 25 , 60 . doi: 10.1186/s13059-024-03198-7 . OpenUrl CrossRef PubMed ↵ Bao , W. , Kojima , K.K. & Kohany , O . ( 2015 ) Repbase update, a database of repetitive elements in eukaryotic genomes . Mobile DNA , 6 , 1 – 6 . doi: 10.1186/s13100-015-0041-9 . OpenUrl CrossRef ↵ Becue , K . ( 2003 ) Ziekten en plagen in prei: deel1 . Landbouw & techniek: tweewekelijks vakblad van de Boerenbond , 8 , 39 – 41 . OpenUrl ↵ Berninger , E. & Buret , P . ( 1967 ) Etude des déficients chlorophylliens chez deux espèces cultivées du genre Allium : l’oignon A. cepa L. et le poireau A. porrum L . Ann. Amelion. Plant , 17 , 175 – 194 . OpenUrl ↵ Bourke , P.M. , Van Geest , G. , Voorrips , R.E. , Jansen , J. , Kranenburg , T. , et al. ( 2018 ) polymapR—linkage analysis and genetic map construction from F1 populations of outcrossing polyploids . Bioinformatics , 34 , 3496 . doi: 10.1093/bioinformatics/bty371 . OpenUrl CrossRef PubMed ↵ Cheng , H. , Concepcion , G.T. , Feng , X. , Zhang , H. & Li , H . ( 2021 ) Haplotype resolved de novo assembly using phased assembly graphs with hifiasm . Nature Methods , 18 , 170 – 175 . doi: 10.1038/s41592-020-01056-5 OpenUrl CrossRef Camacho C. , Coulouris G. , Avagyan V. , Ma N. , Papadopoulos J. , Bealer K. & Madden T.L . ( 2008 ) “ BLAST+: architecture and applications .” BMC Bioinformatics 10 : 421 OpenUrl CrossRef ↵ Danecek , P. , Bonfield , J.K. , Liddle , J. , Marshall , J. , Ohan , V. , et al. ( 2021 ) Twelve years of SAMtools and BCFtools , GigaScience , 10 , 1 – 4 . giab008 , doi: 10.1093/gigascience/giab008 . OpenUrl CrossRef ↵ Dobin , A. , Davis , C.A. , Schlesinger , F. , Drenkow , J. , Zaleski , C. , Jha , S. et al. ( 2013 ) STAR: ultrafast universal RNA-seq aligner . Bioinformatics , 29 , 15 – 21 . doi: 10.1093/bioinformatics/bts635 . OpenUrl CrossRef PubMed Web of Science ↵ Emsweller , S. & Jones H.A . ( 1935a ). Meiosis in Allium fistulosum , Allium cepa , and their hybrid . Hilgardia , 9 , 275 – 294 . DOI: 10.3733/hilg.v09n05p275 . OpenUrl CrossRef ↵ Emsweller , S. & Jones , H.A . ( 1935b ) Gene for control of interstitial localization of chiasmata in Allium fistulosum L . Science , 81 , 543 – 544 . OpenUrl FREE Full Text ↵ Filyushin , M. A. , Beletsky , A. V. , & Kochieva , E. Z . ( 2019 ). Characterization of the complete chloroplast genome of leek Allium porrum L . ( Amaryllidaceae). Mitochondrial DNA Part B , 4 , 2602 – 2603 . doi: 10.1080/23802359.2019.1640090 . OpenUrl CrossRef PubMed ↵ Fu , J. , Zhang , H. , Guo , F. et al. ( 2019 ) Identification and characterization of abundant repetitive sequences in Allium cepa . Sci. Rep ., 9 , 16756 . doi: 10.1038/s41598-019-52995-9 OpenUrl CrossRef PubMed Gabriel , L. , Hoff , K.J. , Brůna , T. et al. ( 2021 ) TSEBRA: transcript selector for BRAKER . BMC Bioinformatics , 22 , 566 . doi: 10.1186/s12859-021-04482-0 OpenUrl CrossRef PubMed ↵ Hao , F. , Liu , X. , Zhou , B. , Tian , Z. , Zhou , L. , Zong , H. , et al. ( 2023 ). Chromosome-level genomes of three key Allium crops and their trait evolution . Nat. Genet ., 55 , 1976 – 1986 . doi: 10.1038/s41588-023-01546-0 . OpenUrl CrossRef PubMed ↵ Hehn , V. ( 1870 ) Kulturpflanzen und Hausthiere in ihrem Übergang aus Asien nach Griechenland und Italien sowie in das übrige Europa: historisch-linguistische Skizzen (Eggers, Ed) Berlin : Gehuder Borntraeger . ↵ Hirschegger , P. , Jakše , J. , Trontelj , P. & Bohanec , B. ( 2010 ). Origins of Allium ampeloprasum horticultural groups and a molecular phylogeny of the section Allium ( Allium: Alliacea e) . Mol. Phyl. Evol . 54 , 488 – 497 . doi: 10.1016/j.ympev.2009.08.030 . OpenUrl CrossRef PubMed Web of Science ↵ Hubley , R. , Finn , R.D. , Clements , J. , Eddy , S.R. , Jones , T.A. , Bao , W. et al. ( 2016 ) The Dfam database of repetitive DNA families . Nucleic Acids Research , 44 , D81 . OpenUrl CrossRef PubMed ↵ Jones , G. , Khazanehdari , K. & Ford-Lloyd , B. ( 1996 ) Meiosis in the leek ( Allium porrum L) revisited .2. Metaphase I observations . Heredity , 76 , 186 – 191 . doi: 10.1038/hdy.1996.26 . OpenUrl CrossRef ↵ Kadry , A. E. R. , Kamel , S. A . ( 1955 ) Cytological studies in the two tetraploid species A. kurrat Schweinf. and A. porrum and their hybrid . Svensk Bot. Tidskr . 49 , 314 – 324 . OpenUrl ↵ Khazanehdari , K.A. & Jones , G.H . ( 1997 ) The causes and consequences of meiotic irregularity in the leek ( Allium ampeloprasum spp. porrum ); implications for fertility, quality and uniformity . Euphytica , 93 , 313 – 319 ; doi: 10.1023/A:1002914808150 . OpenUrl CrossRef Khrustaleva , L.I. , de Melo , P.E. , van Heusden , A.W. & Kik C. ( 2005 ) The integration of recombination and physical maps in a large-genome monocot using haploid genome analysis in a trihybrid Allium population . Genetics , 169 , 1673 – 1685 . doi: 10.1534/genetics.104.038687 . OpenUrl Abstract / FREE Full Text ↵ Kiełkowska , A. ( 2012 ) Meiotic Irregularities in Interspecific Crosses Within Edible Alliums, Meiosis - Molecular Mechanisms and Cytogenetic Diversity , Dr. Andrew Swan (Ed.), ISBN: 978-953-51-0118-5, InTech; http://www.intechopen.com/books/meiosis-molecular-mechanisms-and-cytogenetic-diversity/meiotic-irregularities-in-the-interspecific-crosses-within-edible-Alliums . ↵ Kik , C. , de Groot , L. , Bottema , G. , op ’t Hof , M. , de Visser , N. , Willems , P. , et al. ( 2021 ) Collecting and regenerating populations of the Allium ampeloprasum complex from Greece: Ex situ management of leek crop wild relatives . Gen. Res. , 2 , 1 – 10 . doi: 10.46265/genresj.DMAT2233 . OpenUrl CrossRef ↵ Kirov , I. , Odintsov , S. , Omarov , M. , Gvaramiya , S. , Merkulov , P. , et al. ( 2020 ) Functional Allium fistulosum centromeres comprise arrays of a long satellite repeat, insertions of retrotransposons and chloroplast DNA . Front. Plant Sci ., 11 , 562001 . doi: 10.3389/fpls.2020.562001 . OpenUrl CrossRef PubMed ↵ Kokot , M. , Długosz ,, M. & Deorowicz , S. , ( 2017 ) KMC 3: counting and manipulating k-mer statistics, Bioinformatics , Volume 33 , Issue 17 , Pages 2759 – 2761 , doi: 10.1093/bioinformatics/btx304 OpenUrl CrossRef PubMed ↵ Kudryavtseva , N. , Ermolaev ., A. , Pivovarov ., A. , Simanovsky , S. , Odintsov , S. , et al. ( 2023 ) The control of the crossover localization in Allium . Int. J. Mol. Sci ., 24 , 7066 . doi: 10.3390/ijms24087066 . OpenUrl CrossRef PubMed ↵ Kollmann , F . ( 1972 ) Allium ampeloprasum – a polyploid complex II. Meiosis and relationships between ploidy types . Caryologia , 25 , 295 – 313 . Doi: 10.1080/00087114 ,1972.10796484. OpenUrl CrossRef ↵ Kopecký , D. , Scholten , O. , Majka , J. , Burger-Meijer , K. , Duchoslav , M. & Bartoš , J. ( 2022 ) Genome Dominance in Allium Hybrids ( A. cepa × A. roylei ) . Front. Plant Sci ., 13 , 854127 . doi: 10.3389/fpls.2022.854127 . OpenUrl CrossRef PubMed ↵ Koul A.K. , Gohil R.N . ( 1970 ) Cytology of tetraploid Allium ampeloprasum with chiasma localization . Chromosoma 29 : 12 – 19 . doi: 10.1007/BF01183658 OpenUrl CrossRef ↵ Kovaka , S. , Zimin , A.V. , Pertea , G.M. , Razaghi , R. , Salzberg , S.L. & Pertea , M. ( 2019 ) Transcriptome assembly from long-read RNA-seq alignments with StringTie2 . Genome Biol ., 20 , 278 . doi: 10.1186/S13059-019-1910-1 . OpenUrl CrossRef ↵ Langmead , B. & Salzberg , S . ( 2012 ) Fast gapped-read alignment with Bowtie 2 . Nat. Methods 9 , 357 – 359 . doi: 10.1038/nmeth.1923 . OpenUrl CrossRef PubMed Web of Science Levan , A . ( 1933 ) Cytological studies in Allium IV . Allium fistulosum. Svensk Bot. Tidskr . 27 , 211 – 232 . OpenUrl ↵ Levan , A . ( 1940 ) Meiosis of Allium porrum, a tetraploid species with chiasma localisation . Hereditas 26 , 454 – 462 . OpenUrl ↵ Liao , N. , Hu , Z. , Miao , J. , et al. ( 2022 ) Chromosome-level genome assembly of bunching onion illuminates genome evolution and flavor formation in Allium crops . Nat. Commun . 13 , 6690 . doi: 10.1038/s41467-022-34491-3 . OpenUrl CrossRef ↵ Rabinowitch HD Currah L. Lorbeer , J.W. , Kuhar , T.P. & Hoffmann , M.P. ( 2002 ) Monitoring and forecasting for disease and insect attack in onions and Allium crops within IPM strategies . In: Allium crop science: recent advances . CABI publishing ( Rabinowitch HD , and L. Currah L. eds), pp 293 – 309 . ISBN9780851995106. ↵ Manni , M. , Berkeley , M.R. , Seppey , M. , Simão , F.A. & Zdobnov , E.M. ( 2021a ) BUSCO Update: Novel and Streamlined Workflows along with Broader and Deeper Phylogenetic Coverage for Scoring of Eukaryotic, Prokaryotic, and Viral Genomes . Mol. Biol. Evolution , 38 , 4647 – 4654 . doi: 10.1093/molbev/msab199 . OpenUrl CrossRef PubMed ↵ Manni , M. , Berkeley , M.R. , Seppey , M. & Zdobnov , E.M. ( 2021b ) BUSCO: Assessing Genomic Data Quality and Beyond . Curr. Protoc ., 1 , e323 . doi: 10.1002/CPZ1.323 . OpenUrl CrossRef ↵ Rabinowitch HD Currah L. Mark , G.L. , Gitaitis , G.D. & Lorbeer , J.W. ( 2002 ) Bacterial diseases in onion . In: Allium crop science: recent advances . CABI publishing ( Rabinowitch HD , and L. Currah L. eds), pp 267 – 292 . ISBN9780851995106. ↵ Murín , A . ( 1964 ) Chromosome study in Allium porrum L . Caryologia (Firenze) 17 , 575 – 578 . OpenUrl CrossRef ↵ Nawrocki , E.P. & Eddy , S.R . ( 2013 ) Infernal 1.1: 100-fold faster RNA homology searches , Bioinformatics , Volume 29 , Issue 22 , Pages 2933 – 2935 , doi: 10.1093/bioinformatics/btt509 OpenUrl CrossRef PubMed Web of Science Ohri , D. , Fritsch , R. & Hanelt P . ( 1998 ) Evolution of genome size in Allium ( Alliaceae ) . Plant Syst. Evol ., 210 , 57 – 86 . OpenUrl CrossRef ↵ Pertea , G. & Pertea , M . ( 2020 ) GFF Utilities: GffRead and GffCompare . F1000Res ., 28 , 9 , doi: 10.12688/f1000research.23297.2 . OpenUrl CrossRef ↵ Quinlan , J.R. ( 2014 ) C4.5 Programs for machine learning . Morgan Kaufman Publishers Inc. ISBN 1-55850-238-0. ↵ Ranallo-Benavidez , T.R. , Jaron , K.S. & Schatz , M.C . ( 2020 ) GenomeScope 2.0 and Smudgeplot for reference-free profiling of polyploid genomes . Nat. Commun . 11 , 1432 . doi: 10.1038/s41467-020-14998-3 OpenUrl CrossRef PubMed ↵ Ricroch , A. , Yockteng , R. , Brown , S.C. & Nadot , S . ( 2005 ) Evolution of genome size across some cultivated Allium species . Genome , 48 , 511 – 520 . OpenUrl PubMed ↵ Sahlin , K. & Medvedev , P . ( 2020 ) De Novo clustering of long-read transcriptome data using a greedy, quality balue-Based algorithm . J. Comput. Biol ., 27 , 472 – 484 . doi: 10.1089/cmb.2019.0299 . OpenUrl CrossRef PubMed ↵ Sanderson , H. & Renfrew , J.M. ( 2005 ) The Cultural History of Plants . Routledge . ISBN 0415927463. ↵ Rabinowitch HD Currah L. Salomon ., R. ( 2002 ) Virus diseases in garlic and the propagation of virus-free plants . In: Allium crop science: recent advances . CABI publishing ( Rabinowitch HD , and L. Currah L. eds), pp 311 – 327 . ISBN9780851995106. ↵ Sayers , E. W. , Beck , J. , Bolton , E. E. , Brister , J. R. , et al. ( 2024 ) Database resources of the National Center for Biotechnology Information, Nucleic Acids Research , Volume 52 , Issue D1, 5 Pages D33 – D43 , doi: 10.1093/nar/gkad1044 OpenUrl CrossRef PubMed ↵ Simão , F.A. , Waterhouse , R.M. , Ioannidis , P. , Kriventseva , E.V. & Zdobnov , E.M . ( 2015 ) BUSCO: assessing genome assembly and annotation completeness with single copy-orthologs . Bioinformatics , 31 , 3210 – 3212 . OpenUrl CrossRef PubMed ↵ Smith , A. , Hubley , R. & Green , P. ( 2013 ) RepeatMasker Open-4.0 [2013–2015] . Available from: http://www.repeatmasker.org [Accessed 21st November 2018]. ↵ Stanke , M. , Keller , O. , Gunduz , I. , Hayes , A. , Waack , S. & Morgenstern . B. ( 2006 ) AUGUSTUS: ab initio prediction of alternative transcripts . Nucleic Acids Res ., 34 , W435 – 9 . doi: 10.1093/nar/gkl200 . OpenUrl CrossRef PubMed Web of Science ↵ Stanke , M. , Diekhans , M. , Baertsch , R. , & Haussler , D. ( 2008 ) Using native and syntenically mapped cDNA alignments to improve de novo gene finding . Bioinformatics , 24 , 637 – 644 . doi: 10.1093/bioinformatics/btn013 . OpenUrl CrossRef PubMed Web of Science ↵ Sun , X. , Zhu , S. , Li , N. , Cheng , Y. , Zhao , J. , Qiao , X. , et al. ( 2020 ) A chromosome-level genome assembly of garlic ( Allium sativum ) provides insights into genome evolution and allicin biosynthesis . Mol. Plant , 13 , 1328 – 1339 . OpenUrl CrossRef PubMed ↵ Tang , H. , Zhang , X. , Miao , C. et al. ( 2015 ) ALLMAPS: robust scaffold ordering based on multiple maps . Genome Biol ., 16 , 3 . doi: 10.1186/s13059-014-0573-1 . OpenUrl CrossRef PubMed ↵ Tello , D. , Gil , J. , Loaiza , C.D. , Riascos , J.J. , Cardozo , N. & Jorge Duitama , J . ( 2019 ) NGSEP3: accurate variant calling across species and sequencing protocols , Bioinformatics , 35 , 4716 – 4723 . doi: 10.1093/bioinformatics/btz275 . OpenUrl CrossRef PubMed ↵ Uliano-Silva , M. , Ferreira , J.G.R.N. , Krasheninnikova , K. et al. ( 2023 ) MitoHiFi: a python pipeline for mitochondrial genome assembly from PacBio high fidelity reads . BMC Bioinformatics , 24 , 288 . doi: 10.1186/s12859-023-05385-y . OpenUrl CrossRef PubMed ↵ Voorrips , RE . ( 2002 ) MapChart: software for the graphical presentation of linkage maps and QTLs . J Heredity , 93 , 77 . doi: 10.1093/jhered/93.1.77 . OpenUrl CrossRef PubMed Web of Science ↵ Voorrips , R.E. , Gort , G. & Vosman B . ( 2011 ) Genotype calling in tetraploid species from bi-allelic marker data using mixture models . BMC Bioinformatics , 12 , 172 . doi: 10.1186/1471-2105-12-172 . OpenUrl CrossRef PubMed ↵ Wu , T.D. & Watanabe , C.K . ( 2005 ) GMAP: a genomic mapping and alignment program for mRNA and EST sequences . Bioinformatics , 21 , 1859 – 1875 . doi: 10.1093/bioinformatics/bti310 . OpenUrl CrossRef PubMed Web of Science ↵ Zhou , W. , Armijos , C.E. , Lee , C. , Lu , R. , Wang , J. , et al. ( 2023 ) Plastid Genome Assembly Using Long-read data . Mol. Ecol. Resour ., 23 , 1442 – 1457 . doi: 10.1111/1755-0998.13787 . OpenUrl CrossRef PubMed ↵ Zohary , D. , Hopf , M. & Weiss , E. ( 2012 ) Domestication of Plants in the Old World: The origin and spread of domesticated plants in Southwest Asia, Europe, and the Mediterranean Basin 4th edn. Oxford University Press . doi: 10.1093/acprof:osobl/9780199549061.001.0001 . OpenUrl CrossRef ↵ Zych , K. , Gort , G. , Maliepaard , C.A. , Jansen , R.C. & Voorrips R.E . ( 2019 ) FitTetra 2.0 – improved genotype calling for tetraploids with multiple population and parental data support . BMC Bioinformatics , 20 , 148 . doi: 10.1186/s12859-019-2703-y . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted April 22, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following High-resolution genome and genetic map of tetraploid Allium porrum expose pericentromeric recombination Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share High-resolution genome and genetic map of tetraploid Allium porrum expose pericentromeric recombination Ronald Nieuwenhuis , Roeland Voorrips , Danny Esselink , Thamara Hesselink , Elio Schijlen , Paul Arens , Jan Cordewener , Olga Scholten , Sander Peters bioRxiv 2025.04.21.649809; doi: https://doi.org/10.1101/2025.04.21.649809 Share This Article: Copy Citation Tools High-resolution genome and genetic map of tetraploid Allium porrum expose pericentromeric recombination Ronald Nieuwenhuis , Roeland Voorrips , Danny Esselink , Thamara Hesselink , Elio Schijlen , Paul Arens , Jan Cordewener , Olga Scholten , Sander Peters bioRxiv 2025.04.21.649809; doi: https://doi.org/10.1101/2025.04.21.649809 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Plant Biology Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17691) Bioengineering (13892) Bioinformatics (41937) Biophysics (21452) Cancer Biology (18588) Cell Biology (25504) Clinical Trials (138) Developmental Biology (13378) Ecology (19899) Epidemiology (2067) Evolutionary Biology (24320) Genetics (15609) Genomics (22506) Immunology (17736) Microbiology (40394) Molecular Biology (17181) Neuroscience (88605) Paleontology (666) Pathology (2832) Pharmacology and Toxicology (4824) Physiology (7641) Plant Biology (15156) Scientific Communication and Education (2045) Synthetic Biology (4294) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-06-02T02:00:03.124865+00:00
License: CC-BY-NC-ND-4.0