Full text
34,034 characters
· extracted from
preprint-html
· click to expand
DipGNNome: Diploid de novo genome assembly with geometric deep learning and beam-search | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results DipGNNome: Diploid de novo genome assembly with geometric deep learning and beam-search View ORCID Profile Martin Schmitz , View ORCID Profile Lovro Vrček , Kenji Kawaguchi , View ORCID Profile Mile Šikić doi: https://doi.org/10.1101/2025.09.16.676474 Martin Schmitz 1 School of Computing, National University of Singapore , Singapore 2 Genome Institute of Singapore , A*STAR, Singapore Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Martin Schmitz For correspondence: schmitz.kessenich{at}gmail.com Lovro Vrček 2 Genome Institute of Singapore , A*STAR, Singapore 3 Faculty of Electrical Engineering and Computing, University of Zagreb , Croatia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Lovro Vrček Kenji Kawaguchi 1 School of Computing, National University of Singapore , Singapore Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mile Šikić 2 Genome Institute of Singapore , A*STAR, Singapore 3 Faculty of Electrical Engineering and Computing, University of Zagreb , Croatia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Mile Šikić Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract De novo genome assembly remains a central challenge in computational biology, particularly for diploid genomes where maternal and paternal haplotypes must be accurately resolved. Existing assemblers achieve impressive results through carefully designed heuristics, yet modern deep learning methods remain largely unexplored in the diploid setting. We present DipGNNome , the first deep learning–based framework for diploid de novo genome assembly. Our approach formulates assembly as an edge classification and traversal problem on haplotype-aware assembly graphs, training graph neural networks (GNNs) to guide contig construction. To enable this, we establish a novel pipeline for generating diploid graphs with ground-truth edge labels, providing the first systematic way to produce training data for machine learning models in this domain. This framework creates a foundation for applying and extending graph-based deep learning to diploid assembly. DipGNNome creates assemblies comparable to state-of-the-art and demonstrates the feasibility of deep learning for diploid assembly and introduces a paradigm that bridges algorithmic genomics with graph representation learning. Our code, dataset and trained model is openly available at https://github.com/lbcb-sci/DipGNNome . 1 Introduction The genome is encoded in sequences of four nucleotides—adenine (A), guanine (G), cytosine (C), and thymine (T)—that together form deoxyribonucleic acid (DNA). DNA is organized into chromosomes, each spanning millions to hundreds of millions of nucleotides. A genome typically consists of multiple chromosomes, often present in pairs or higher ploidy levels, where homologous chromosomes are similar but not identical. Sequencing technologies make it possible to extract genomic information from these chromosomes, producing large collections of short, unordered fragments known as reads. Reconstructing the full genome from such reads is the central problem of genome assembly. Traditional assembly methods align reads to a reference genome. While effective when a high-quality reference is available, this approach can introduce bias and obscure genuine structural variation between the target and reference genomes. In contrast, de novo assembly reconstructs genomes directly from reads, without relying on a reference, thereby avoiding these biases. De novo assembly is a cornerstone of computational genomics, with applications spanning foundational research in biology, evolutionary biology, and personalized medicine. Nurk et al. (2022) [ 12 ] produced the first complete telomere-to-telomere (T2T) human genome, followed by numerous high-quality assemblies for humans and other species. These efforts combine complementary sequencing technologies, sophisticated assembly algorithms, and extensive manual curation by large consortia. Progress of recent assemblers has been driven primarily by algorithmic innovations. State-of-the-art assemblers like hifiasm [ 3 ] and Verkko [ 14 ] exploit the high accuracy of PacBio HiFi reads and the ultra-long range of Oxford Nanopore reads to generate phased, near-complete T2T assemblies. Yet, despite these achievements, current approaches remain purely algorithmic and do not incorporate modern machine learning techniques that have transformed other domains. In this work, we present DipGNNome , the first deep learning–based assembler capable of reconstructing diploid genomes directly from raw assembly graphs, augmented with trio information. Unlike existing methods, DipGNNome integrates GNNs with haplotype-aware graph processing to jointly reconstruct maternal and paternal haplotypes, bridging the gap between classical algorithmic assemblers and emerging machine learning approaches. 2 Related Work GNNome [ 18 ] marks the first deep learning–based genome assembler. It trains a GNN to classify assembly graph edges as correct or incorrect and then applies a greedy pathfinding algorithm to construct contigs from the scored edges. Similarly, Simunovic et al. [ 15 ] focus on the layout of De Bruijn graphs. However, these approaches are limited to producing haploid assemblies. Diploid assembly poses additional challenges. Most organisms, including humans, carry two homologous copies of each chromosome. While homozygous regions are nearly identical, heterozygous regions can harbor substantial divergence. Distinguishing sequencing errors from true variants and correctly phasing both haplotypes remains central difficulties in diploid assembly. Recent years have seen the development of deep learning–based phasing methods such as GAEseq [ 6 ], CAECseq [ 5 ], NeurHap [ 19 ], and ralphi [ 1 ]. These approaches highlight the potential of machine learning for haplotype resolution, but they operate in a reference-based setting and do not address de novo assembly. Traditional strategies for de novo diploid assembly frequently incorporate external information to resolve haplotypes. A prominent example is trio binning [ 7 ], which leverages parental short reads to identify haplotype-specific k-mers and guide separation during assembly, as implemented in tools such as Verkko [ 14 ] and hifiasm [ 3 ]. Our method builds on this paradigm by combining haplotype-aware assembly graphs with parental k-mer annotation and graph-based machine learning. 3 Overview Our contributions are threefold: (1) we introduce the first publicly available pipeline for constructing machine learning–ready assembly graphs with ground-truth, enabling supervised training in the diploid setting, and share our dataset; (2) we develop an assembly algorithm that combines model predictions with a beam-search strategy to efficiently traverse long, string-like graphs with limited branching, a design that may generalize to other path-finding tasks; and (3) we show that a GNN-based assembler can achieve competitive accuracy, rivaling state-of-the-art methods while following a fundamentally different, learning-driven paradigm. Figure 1 outlines the three stages of our approach: (A) Data processing , where HiFi reads are assembled into a unitig graph with hifiasm , simplified, and annotated with haplotype information using trio-derived k -mers; (B) Synthetic training , where simulated diploid reads generate labeled graphs for supervised GNN training via edge classification; and (C) Genome assembly , where real data are processed as in A, edges are scored by the trained GNN, and a beam-search reconstructs phased maternal and paternal haplotypes. Download figure Open in new tab Fig. 1. Overview of the DipGNNome pipeline. A: Data processing . HiFi reads are assembled into a unitig graph with hifiasm , simplified, and annotated with haplo-type information using trio-derived k -mers. B: Synthetic training data . Simulated diploid reads generate labeled graphs, enabling supervised training of a GNN for edge classification. C: Genome assembly . Real data is processed as described in A , edges are scored by the trained GNN from B , and a beam-search based algorithm reconstructs phased haplotypes. The remainder of the paper is organized as follows: The next three sections describe the parts of the pipeline, (A), (B) and (C). Section 4 details the data processing pipeline, Section 5 describes GNN training, and Section 6 presents the genome assembly algorithm. Section 7 evaluates DipGNNome on real and simulated datasets, comparing it to state-of-the-art assemblers. Section 8 concludes with a discussion of implications, limitations, and future directions. 4 Data Processing Pipeline The following section describes how we construct graphs as summarized in Figure A . Given a set of reads (real or synthetic) as input, the pipeline consists of the following steps: 4.1 Create raw Overlap Graph (Step 1) We start with an overlap graph, where each node represents a read in the dataset and each edge represents an overlap. Any tool that can produce such a graph is possible to be used here. For our experiments, we use hifiasm v0.25 [ 4 ] with default parameters. 4.2 Convert raw Graph to Unitig Graph (Step 2) Given the raw overlap graph, we search for transitive edges using an algorithm based on Myers’ transitive edge reduction algorithm [ 11 ]. An edge e t = a → c is considered transitive if edges e 1 = a → b and e 2 = b → c both exist. To relax this rule, we introduce an ϵ= 0.12 parameter (inspired by Raven [ 17 ]), which allows transitive edges to be removed only if the indirect path length lies between d (1 — ϵ) and d(1 + ϵ) (where d is the direct path). After the removal of transitive edges, unbranching chains of nodes are merged into so-called unitigs . This significantly reduces the size and complexity of the graph. The resulting graph is called a unitig graph. 4.3 Add Trio Binning to Unitigs using yak (Step 3) We use Illumina short reads from both parents of the genome of interest and apply yak [ 4 ] trio binning. yak identifies unique k-mers in both maternal and paternal short reads. Then it traverses the unitigs in the graph and marks unique k-mers as identified before. The output is a file with unique k-mer hits for each unitig. We leverage the counts of unique k-mer hits, k m ( v ) and k p ( v ). These values provide an informative representation of each unitig, capturing both the haplotype assignment (maternal vs. paternal) and the strength of the heterozygosity (the absolute count k m ( v ) and k p ( v ) may reflect how heterozygous a unitig is). 4.4 Add Node and Edge Features (Step 4) We enrich the graph with both node and edge features. Edge features include the overlap length (number of base pairs shared between two reads). Node features include read support (the number of raw reads contributing to a given unitig, with read-level coverage computed by hifiasm and aggregated in unitigs as a length-weighted average) as well as in-degree/out-degree (the number of incoming and outgoing edges, capturing a node’s connectivity within the graph). 5 GNN Training This Section explains the GNN training pipeline as shown in Figure 1 B . Given a set of reference genomes, we create a synthetic training dataset with ground-truth and use this to train a GNN. 5.1 Sample reads We simulate 40× coverage reads (20× per haplotype) using PBSim3 [ 13 ], sampling independently from maternal and paternal references of each chromosome. The resulting FASTQ files are merged into one, with reads annotated by position, strand, chromosome, and haplotype. 5.2 Compute Ground-Truth Training our GNN requires labeled assembly graph edges. We first mark all edges connecting different chromosomes (translocations) or opposite strands (inversions) as false. Translocation edges are additionally saved as a separate list to train a second translocation classifier. Positional information determines whether an edge represents a valid suffix–prefix overlap. This requires assigning coordinates to unitigs. For diploid genomes, differences in haplotype coordinates between homologous loci complicate verification. To resolve this, we translate coordinates between haplotypes using Liftover [ 16 ] and refine missing values by leveraging information from multiple reads within unitigs. Overlaps are then validated in both coordinate systems, and an edge is labeled correct if it is consistent in at least one haplotype. This procedure yields reliable supervision signals for robust model training. A detailed description of the coordinate translation and ground-truth verification algorithm is provided in Appendix A. 5.3 GNN Model We train a SymGatedGCN model with SymBCELoss both introduced by GN-Nome [ 18 ], using the ground-truth edge labels described above. SymGatedGCN extends GatedGCN [ 2 ], generating d -dimensional node and edge embeddings through double message passing to capture edge directionality. PairNorm [ 22 ] is applied after each layer to prevent oversmoothing. The model and its parts are described in more detail in Appendix B. Node features and edge features are first projected into d -dimensional embeddings through the GNN layers. A final MLP classifier, trained jointly with the GNN, combines the edge, source node, and target node embeddings to predict two scores: the general edge score and the probability of a translocation edge. These predictions guide both graph pruning and downstream diploid genome assembly. Predictions are optimized against ground-truth using binary cross-entropy loss. 6 Genome Assembly Algorithm This section corresponds to part C of Figure 1 and describes how we use the trained GNN from Section 5 together with the processed graphs from Section 4 to construct a dual genome assembly. The result consists of two independent sets of contigs, one representing the maternal haplotype and the other the paternal haplotype. Problem Formulation We formulate assembly as a path-finding problem in a directed graph G = ( V, E ) where nodes represent unitigs and edges represent overlaps. Each node is associated with its sequence length and haplotype-specific k-mer counts, while edges carry model-predicted quality scores and overlap lengths. The graph is decomposed into weakly connected components, and low-quality and translocation-predicted edges are pruned before assembly. Assembly Strategy Assembly proceeds independently for the maternal and paternal haplotypes. For each haplotype, we iteratively traverse graph components, extracting one contig at a time until no sufficiently long path remains. Paths are initiated from sampled edges and constructed using a diploid-adapted beam-search. Each run performs a forward search from one node and a reverse search from the complement, which are then combined into a candidate contig. Among multiple candidates, the longest is retained. Beam-Search Adaptations Our AssemblyBeamSearch algorithm extends standard beam-search with several modifications tailored to path finding on a string-like assembly graph: (1) Best-beam tracking ensures high-scoring partial contigs are not lost even if only negative edges remain. (2) Complement enforcement prevents a node and its reverse complement from appearing in the same contig. (3) Beam merging removes redundant paths when multiple beams converge on the same node, improving runtime and memory efficiency. Together, these modifications balance contiguity with haplotype consistency. Scoring Function Each candidate edge is evaluated by a composite score that rewards sequence length while penalizing haplotype conflicts and low-quality connections: where l ( v ) is node length, l ( e ) is overlap, k ¬ h ( v ) counts unique k -mers from the opposite haplotype, and m ( e ) is the GNN-predicted edge error. Hyperparameters α, β , γ control the trade-off. In essence, our algorithm constructs haplotype-specific assemblies by iteratively applying a modified beam-search, guided by model predictions, haplotype markers, and sequence length. Full algorithmic details and pseudocode, as well as the derivation of the beam-search heuristic, are provided in Appendix C. 7 Evaluation We evaluate our methods’ capacity on various real genomes. Our model is trained on synthetic human data based on a single reference genome: I002C [ 8 ]. For evaluation, we focus exclusively on real-world data and assess assembly quality across a range of datasets consisting of a different human individual and 6 primate species. All experiments are conducted at the full diploid genome level, using datasets for which high-quality reference genomes are available. 7.1 Training Dataset Overview We construct a training dataset, consisting of graphs generated from all chromosomes of the human genome I002C, with about 7.7M nodes and 10.5M edges across training and validation splits, and an average degree of 2.7. Roughly 85% of edges are labeled as correct, while the remaining 15% correspond to false edges. A detailed breakdown of the training dataset statistics is provided in Appendix D. 7.2 Evaluation Dataset Overview The evaluation set comprises seven genomes: The human ( H. sapiens ) reference HG002 by Nurk et al. 2022 [ 12 ], and the six T2T ape references created by Yoo et al. 2024 [ 21 ]: P. paniscus (bonobo), G. gorilla (gorilla), P. troglodytes (chimpanzee), S. syndactylus (siamang), P. abelii (Sumatran orangutan), and P. pygmaeus (Bornean orangutan). For P. troglodytes, S. syndactylus, P. abelii , and P. pygmaeus , no parental short-read data were available. Instead, we generated unique k-mers directly from the reference genomes. The exact filesand coverages used for the evaluation dataset are detailed in Appendix 7.3 Training Setup To allow for mini-batch training on large graphs, we partition each graph using METIS into subgraphs containing at most 40,000 nodes (which is approximately the graph size to fit the full GPU memory). Each resulting subgraph is treated as a training batch. More training details such as the final model configuration and the hardware used can be found in Appendix F. 7.4 Genome Assembly Results We benchmark three approaches: a greedy variant of DipGNNome , the full DipGNNome algorithm, and hifiasm . The greedy variant is identical to DipGNNome except that it replaces beam-search with a purely greedy expansion strategy, always selecting the next best node (according to the same scoring metric as the original algorithm). Table 1 shows the results including assembly length, duplication rate (Rdup), contiguity (NG50 and NGA50), and haplotype error rates (YAK Switch Error and Hamming Error) for paternal (P) and maternal (M) haplotypes. Values are shown as P/M (paternal/maternal). Hamming and switch error rates were computed using yak [ 4 ], while all other metrics (length, duplication rate, NG50, NGA50) were computed using minigraph [ 9 ]. Lower duplication and error rates indicate better assembly quality, while higher NGA50 reflects better (correct) contiguity. NG50 and total length alone do not necessarily indicate better or worse assembly quality. However, differences between NG50 and NGA50 display misassemblies. View this table: View inline View popup Download powerpoint Table 1. Comparison of DipGNNome and hifiasm on different genomes. Across most genomes, hifiasm achieves the lowest Hamming error rates and remains the strongest assembler overall. Nonetheless, DipGNNome performs competitively, often reaching comparable NGA50 values and in some cases even surpassing hifiasm. In particular, the P. pygmaeus assemblies exhibit substantially higher NGA50 with DipGNNome , while maintaining switch and Hamming errors close to those of hifiasm for both haplotypes. For S. syndactylus (M) and P. abelii (P), DipGNNome also achieves notable NGA50 improvements, albeit with a modest increase in Hamming error. Across all other haplotypes, however, hifiasm remains superior to or on par with DipGNNome . The greedy variant consistently performs worse, with lower NGA50 and higher error rates, confirming the importance of our beam-search based algorithm for robust path exploration. 8 Conclusion This work presents the first successful application of deep learning to diploid de novo genome assembly, demonstrating that GNNs can effectively resolve complex assembly graphs. Building upon concepts from GNNome [ 18 ], we introduce a series of optimizations and novel contributions in both the data processing pipeline and the assembly algorithm. We establish the first framework for generating diploid graphs with ground-truth edge correctness, when read coordinates originate from distinct reference systems for the same loci. This provides a foundation for creating diploid graphs in DGL or PyG format that can be readily applied to train machine learning models for diploid genome assembly. We further present a beam-search based path-finding algorithm that incorporates new strategies, such as beam merging, which proves particularly effective for navigating string-like graph structures. We can roughly match State-of-the-Art (SotA) results for most genomes, and for P. pygmaeus clearly outperform SotA. For the other genomes DipGNNome achieves contiguity and switch error rates comparable to hifiasm , although it exhibits higher Hamming error. A likely explanation is the absence of any polishing step in our pipeline. Unlike hifiasm , which integrates several read-level polishing operations directly into the assembly process, our method currently performs no equivalent refinement. Our approach identifies paths on the raw graph in a Hamiltonian-like manner, without reusing nodes. In contrast, hifiasm occasionally reuses nodes in its assembly and introduces additional connections beyond those present in the raw graph, in addition to its built-in read-level polishing steps. It is not clear whether further improvements can be achieved from path-finding alone, without incorporating such steps. Another factor influencing performance is the diversity of the training data. Our current model is trained exclusively on human samples, which possibly limits its generalization to more divergent species such as plants. Nevertheless, we demonstrate successful transfer to other primates. This limitation is not fundamental: DipGNNome can be retrained with data from a broader range of species and sequencing technologies, making it adaptable both to new organisms and to future advances in sequencing. Several challenges remain. Most importantly, DipGNNome is constrained by the quality of the input graphs. The effectiveness of the layout phase depends on the completeness of the overlap graph, which in our case is generated by hifiasm . Missing edges can lead to fragmentation or the absence of valid end-to-end paths, since our GNN-based method can only select among existing edges—it cannot create new ones. This fundamentally limits performance. Because we rely on the same input graphs as hifiasm , it is inherently difficult to surpass it: errors introduced during hifiasm ‘s error correction and graph construction propagate directly into our pipeline. Overcoming this would require designing a graph generation algorithm tailored to GNN-based assembly. Such graphs could deliberately include more edges (accepting some incorrect ones) to provide the model with greater flexibility. Whereas hifiasm minimizes incorrect edges at the cost of reduced connectivity, our approach may benefit from denser graphs, as the GNN can learn to identify and remove spurious links. Moreover, our pipeline can directly integrate graphs from other assemblers by replacing the hifiasm -based graph generation stage and retraining the model accordingly. References 1. ↵ Battistella , E. , Maheshwari , A. , Ekim , B. , Berger , B. , Popic , V. : ralphi: a deep reinforcement learning framework for haplotype assembly . In: International Conference on Research in Computational Molecular Biology . pp. 349 – 353 . Springer ( 2025 ) 2. ↵ Bresson , X. , Laurent , T. : Residual gated graph convnets . arXiv preprint arxiv: 1711.07553 ( 2017 ) 3. ↵ Cheng , H. , Asri , M. , Lucas , J. , Koren , S. , Li , H. : Scalable telomere-to-telomere assembly for diploid and polyploid genomes with double graph . Nature Methods pp. 1 – 4 ( 2024 ) 4. ↵ Cheng , H. , Concepcion , G.T. , Feng , X. , Zhang , H. , Li , H. : Haplotype-resolved de novo assembly using phased assembly graphs with hifiasm . Nature methods 18 ( 2 ), 170 – 175 ( 2021 ) OpenUrl PubMed 5. ↵ Ke , Z. , Vikalo , H. : A convolutional auto-encoder for haplotype assembly and viral quasispecies reconstruction . Advances in Neural Information Processing Systems 33 , 13493 – 13503 ( 2020 ) OpenUrl 6. ↵ Ke , Z. , Vikalo , H. : A graph auto-encoder for haplotype assembly and viral quasispecies reconstruction . In: Proceedings of the AAAI Conference on Artificial Intelligence . vol. 34 , pp. 719 – 726 ( 2020 ) OpenUrl 7. ↵ Koren , S. , Rhie , A. , Walenz , B.P. , Dilthey , A.T. , Bickhart , D.M. , Kingan , S.B. , Hiendleder , S. , Williams , J.L. , Smith , T.P. , Phillippy , A.M. : De novo assembly of haplotype-resolved genomes with trio binning . Nature biotechnology 36 ( 12 ), 1174 – 1182 ( 2018 ) OpenUrl CrossRef 8. ↵ LBCB-SCI: Telomere-to-telom https://github.com/lbcb-sci/I002C (2024), https://github.com/lbcb-sci/I002C ( 2024 ), 28.12.2024 9. ↵ Li , H. , Feng , X. , Chu , C. : The design and construction of reference pangenome graphs with minigraph . Genome biology 21 , 1 – 19 ( 2020 ) OpenUrl CrossRef PubMed 10. Luu , P.L. , Ong , P.T. , Dinh , T.P. , Clark , S.J. : Benchmark study comparing liftover tools for genome conversion of epigenome sequencing data . NAR genomics and bioinformatics 2 ( 3 ), qaa054 ( 2020 ) 11. ↵ Myers , E.W. : The fragment assembly string graph . Bioinformatics 21 ( Suppl_2 ), ii79 – ii85 ( 2005 ) OpenUrl CrossRef PubMed 12. ↵ Nurk , S. , Koren , S. , Rhie , A. , Rautiainen , M. , Bzikadze , A.V. , Mikheenko , A. , Vollger , M.R. , Altemose , N. , Uralsky , L. , Gershman , A. , et al : The complete sequence of a human genome . Science 376 ( 6588 ), 44 – 53 ( 2022 ) OpenUrl CrossRef PubMed 13. ↵ Ono , Y. , Asai , K. , Hamada , M. : Pbsim: Pacbio reads simulator—toward accurate genome assembly . Bioinformatics 29 ( 1 ), 119 – 121 ( 2013 ) OpenUrl CrossRef PubMed Web of Science 14. ↵ Rautiainen , M. , Nurk , S. , Walenz , B.P. , Logsdon , G.A. , Porubsky , D. , Rhie , A. , Eichler , E.E. , Phillippy , A.M. , Koren , S. : Telomere-to-telomere assembly of diploid chromosomes with verkko . Nature Biotechnology 41 ( 10 ), 1474 – 1482 ( 2023 ) OpenUrl CrossRef PubMed 15. ↵ Šimunovic , M. , Šikic , M. , Bankevich , A. : Gnndebugger: Gnn based error correction in de bruijn graphs . bioRxiv pp. 2025 – 05 ( 2025 ) 16. ↵ Talenti , A. , Prendergast , J. : nf-lo: a scalable, containerized workflow for genometo-genome lift over . Genome Biology and Evolution 13 ( 9 ), evab183 ( 2021 ) OpenUrl CrossRef PubMed 17. ↵ Vaser , R. , Šikic , M. : Time-and memory-e?cient genome assembly with raven . Nature Computational Science 1 ( 5 ), 332 – 336 ( 2021 ) OpenUrl PubMed 18. ↵ Vrcek , L. , Bresson , X. , Laurent , T. , Schmitz , M. , Kawaguchi , K. , Šikic , M. : Geometric deep learning framework for de novo genome assembly . Genome research 35 ( 4 ), 839 – 849 ( 2025 ) OpenUrl Abstract / FREE Full Text 19. ↵ Xue , H. , Rajan , V. , Lin , Y. : Graph coloring via neural networks for haplotype assembly and viral quasispecies reconstruction . Advances in Neural Information Processing Systems 35 , 30898 – 30910 ( 2022 ) OpenUrl 20. Yang , C. , Zhou , Y. , Song , Y. , Wu , D. , Zeng , Y. , Nie , L. , Liu , P. , Zhang , S. , Chen , G. , Xu , J. , et al : The complete and fully-phased diploid genome of a male han chinese . Cell Research 33 ( 10 ), 745 – 761 ( 2023 ) OpenUrl CrossRef PubMed 21. ↵ Yoo , D. , Rhie , A. , Hebbar , P. , Antonacci , F. , Logsdon , G.A. , Solar , S.J. , Antipov , D. , Pickett , B.D. , Safonova , Y. , Montinaro , F. , et al : Complete sequencing of ape genomes . bioRxiv pp. 2024 – 07 ( 2024 ) 22. ↵ Zhao , L. , Akoglu , L. : Pairnorm: Tackling oversmoothing in gnns . In: International Conference on Learning Representations ( 2020 ), https://openreview.net/forum?id=rkecl1rtwB View the discussion thread. Back to top Previous Next Posted September 18, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following DipGNNome: Diploid de novo genome assembly with geometric deep learning and beam-search Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share DipGNNome: Diploid de novo genome assembly with geometric deep learning and beam-search Martin Schmitz , Lovro Vrček , Kenji Kawaguchi , Mile Šikić bioRxiv 2025.09.16.676474; doi: https://doi.org/10.1101/2025.09.16.676474 Share This Article: Copy Citation Tools DipGNNome: Diploid de novo genome assembly with geometric deep learning and beam-search Martin Schmitz , Lovro Vrček , Kenji Kawaguchi , Mile Šikić bioRxiv 2025.09.16.676474; doi: https://doi.org/10.1101/2025.09.16.676474 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genomics Subject Areas All Articles Animal Behavior and Cognition (7618) Biochemistry (17636) Bioengineering (13860) Bioinformatics (41847) Biophysics (21401) Cancer Biology (18536) Cell Biology (25424) Clinical Trials (138) Developmental Biology (13353) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24287) Genetics (15583) Genomics (22463) Immunology (17701) Microbiology (40300) Molecular Biology (17141) Neuroscience (88434) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4813) Physiology (7633) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4285) Systems Biology (9808) Zoology (2268)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.