ANI-netID: a genome similarity-network based biological identification system

preprint OA: closed CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Fungal identification is based on sequence similarity comparisons, with multiple genes often utilized in conjunction, similar to most taxonomic studies. The identification process involves the individual comparison of the similarities of each gene. However, owing to insufficient information, incomplete lineage sorting, gene transfer, hybridization, gene duplication, and loss, different species may match, depending on the DNA region used for comparison. Additionally, identification methods exist that set a uniform similarity threshold value, but they hit multiple species above the threshold. In this study, we introduced ANI-netID, which was developed to address these challenges. ANI-netID is an identification method based on the phylogenetic species concept, where the threshold value is determined by the combination of individuals with the lowest nucleotide similarity within groups (clique groups), such as species and genera. This checks whether the queried individual fall into the same clique group as the reference individual with the highest similarity. ANI-netID solves not only the above-mentioned problems but also indicates that the individual being identified may represent a new lineage, if the individual does not belong to any clique group.
Full text 32,858 characters · extracted from preprint-html · click to expand
ANI-netID: a genome similarity-network based biological identification system | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results ANI-netID: a genome similarity-network based biological identification system S. Nozawa , View ORCID Profile K. Watanabe doi: https://doi.org/10.1101/2025.04.19.649685 S. Nozawa 1 School of Agriculture, Tamagawa University, Machida , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: nozao802{at}gmail.com wkyoko{at}agr.tamagawa.ac.jp K. Watanabe 1 School of Agriculture, Tamagawa University, Machida , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for K. Watanabe For correspondence: nozao802{at}gmail.com wkyoko{at}agr.tamagawa.ac.jp Abstract Full Text Info/History Metrics Preview PDF Abstract Fungal identification is based on sequence similarity comparisons, with multiple genes often utilized in conjunction, similar to most taxonomic studies. The identification process involves the individual comparison of the similarities of each gene. However, owing to insufficient information, incomplete lineage sorting, gene transfer, hybridization, gene duplication, and loss, different species may match, depending on the DNA region used for comparison. Additionally, identification methods exist that set a uniform similarity threshold value, but they hit multiple species above the threshold. In this study, we introduced ANI-netID, which was developed to address these challenges. ANI-netID is an identification method based on the phylogenetic species concept, where the threshold value is determined by the combination of individuals with the lowest nucleotide similarity within groups (clique groups), such as species and genera. This checks whether the queried individual fall into the same clique group as the reference individual with the highest similarity. ANI-netID solves not only the above-mentioned problems but also indicates that the individual being identified may represent a new lineage, if the individual does not belong to any clique group. Introduction Species identification is important in biodiversity research and exerts a significant influence on the field of applied biological research. The 1980s witnessed significant advancements in the field, with the development of methods for determining the base sequences of genes. Concurrently, several databases have been established to facilitate the storage and exchange of this information, including GenBank (National Center for Biotechnology Information base sequence database), DNA Data Bank of Japan (DDBJ), and the European Nucleotide Archive (ENA) of the European Molecular Biology Laboratory (EMBL). Currently, a substantial corpus of nucleotide sequence data has been amassed and is being used for identification purposes. In the context of research on microorganisms that possess a paucity of distinguishing characteristics, such as fungi, DNA-based identification methods have proven to be highly advantageous. These methods are frequently employed in the identification of pathogens that affect humans, animals, and agricultural crops [ 1 ]. For the identification of fungi, the Basic Local Alignment Search Tool (BLAST) selects the species name with the highest percentage of similarity. However, the absence of a predetermined similarity threshold for selecting the species name makes it impossible to account for the potential of the identified fungus as a novel lineage. Conversely, recent methodologies have employed a threshold for identification, particularly in metabarcoding analyses. For instance, tools for species identification based on ITS sequences have been developed, such as CloVR-ITS, DAnIEL, and PIPITS [ 2 – 4 ]. These tools determine the species name when there are species that match the threshold value (e.g., 97% similarity) or higher for species identification. Cai and Druzhinina (2021) [ 5 ] published criteria for the identification of Trichoderma species based on a threshold for species identification based on similarity values. Specifically, they stated that the reference strains that match 99% or more using the second largest subunit of RNA polymerase II as an indicator or 97% or more using the translation elongation factor 1-alpha gene as an indicator, are the same species. However, some combinations demonstrated a similarity value of 99% or more between species. Consequently, the efficacy of species identification based on uniform similarity values is contingent upon the specific strain in question. Another problem with the current species identification technique is that because species classification is currently based on multiple DNA regions in several taxa, multiple regions must also be used for species identification. This means that each region is individually analyzed using BLAST to identify species with high similarity. However, when species identification is based on multiple DNA regions, different species are often hit most similarly, depending on the region analyzed. This is due to the evolutionary history of each gene, including incomplete lineage sorting, hybridization in ancestral lineages, gene transfer, gene duplication, and loss [ 6 ]. In this case, identification cannot be based on similarity comparisons. Therefore, species must be identified by molecular phylogenetic analysis based on data combining DNA regions according to the latest taxonomic studies. First, in the context of the current taxonomy, in which multiple gene regions are combined for phylogenetic analysis, it is not surprising that discrepancies arise in the identification results obtained from the similarity analyses of individual genes. In this study, we propose a novel identification system called the Average Nucleotide Identity Network Identification (ANI-NetID). This biological identification system is based on groups (clique groups) composed of individuals whose sequences have been annotated (e.g., groups composed of individuals of the same species). This checks whether the queried individual belongs to the same clique group as the reference individual with the highest similarity. This system has the advantage of eliminating subjectivity because the threshold values are set according to the rules. Furthermore, it can be concluded that individuals who do not belong to any clique may belong to a new lineage. Additionally, when employing multiple gene regions in this network, the threshold is determined based on the Average Nucleotide Identity, facilitating identification. This system solves the problem of identifying different species for each gene. In this study, we demonstrated ANI-NetID through simulations using a dataset to identify the species of Nigrospora fungi. Materials and Methods 1. ANI-NetID pipeline 1) The reference sequence data are integrated with the query sequence for each DNA region. 2) Alignment is performed independently for each DNA region; ANI-netID uses MAFFT v7.310 [ 7 ] for multiple alignments, which creates a guide tree based on distances and aligns sequences based on this tree, which is compatible with current classification methods that build classification systems based on phylogenetic trees. Note that BLAST, which finds pairwise similarities without creating a guide tree, is not used in the next process of obtaining similarity values. 3) After alignment, the similarity values between sequences for each DNA region are determined using exhaustive pairwise sequence comparisons. 4) The average similarity of the DNA regions is obtained for all individual combinations. For example, if the homology of each region between two individuals is Locus A: 97%, Locus B: 98%, and Locus C: 99%, the average is 98. 5) The pair with the lowest similarity within each clique group is determined based on the average similarity, and its similarity value (LAC; lowest ANI value in the clique group) is obtained. 6) We then search for the reference sequence with the highest similarity to the query sequence and obtain its similarity value (HAQ: highest ANI value to the query). 7) HAQ is compared with the LAC among the combinations of individuals in the clique group to which the individual belongs and is identified based on the following identification evaluation criteria: HAQ =100%: It is identified as belonging to the same clique group as an individual if it is 100% similar to a query. HAQ > LAC: The query belongs to the clique group with the most similar individuals. In this case, the query is identified by a clique group. HAQ≦LAC: A query is closely related to the clique group to which the most similar individuals belong, yet it does not belong to either clique group. In this case, as a phylogenetic relationship, a query may be located in the ancestral lineage of the clique group or in an independent lineage. Consequently, queries should be assessed using molecular phylogenetic analyses. The program developed in this study (ANInetID.py; S1 Dataset) produced the following final outputs: i) the name of the reference individual most homologous to the query, ii) its similarity value (HAQ), iii) the name of the clique group to which the query belongs, and iv) the threshold value of each clique group (LAC). By truncating the results to an arbitrary value, the identification results can be visualized by depicting a network diagram in which the individuals above that value are connected. Moreover, the output of the identification process, encompassing the alignment results of the DNA regions, respective homology values, and mean values of the DNA regions utilized, can be obtained. 3. Experiment Simulations were performed using the Python program (ANInetID.py; S1 Dataset) developed in this study based on a biological dataset. The taxa targeted in the simulations were Nigrospora spp. Five query sequences (query 1-5) were identified using sequences of 58 strains from 29 species ( Table 1 ) that had already been identified as reference sequences. The DNA regions used as reference were the three regions the internal transcribed spacer region (ITS), transcribed elongation factor 1 alpha ( tef1 ), and β-tubulin , which have been used in taxonomic studies of the genus. The reference sequences in each region were trimmed at both ends after multiple alignments in advance, so that the common 5‘ to 3’ end sites throughout the sequence could be identified as the reference. No individuals from the other clique groups were mixed within the clique group (S1 Fig). View this table: View inline View popup Table 1. GenBank accession numbers of the DNA sequences used in this study. View this table: View inline View popup Table 2. Pairs with lowest identities in each clique group (species) and their identity Phylogenetic analyses Phylogenetic analyses were performed to evaluate the identification results obtained using ANI-NetID. To construct phylogenetic trees, three query strains and 59 strains of Negrospora spp., including ex-type strains, which were from the same dataset as that for ANI-NetID, were used ( Table 1 ). As out groups, Apiospora malaysiana and A. pseudoparenchymatica were used. All the ITS, β-tubulin , and translation elongation factor 1-alpha gene ( tef1 ) sequences were used for multiple alignments by MAFFT v7.310 and were concatenated into a dataset. Three phylogenetic analyses—neighbor-joining (NJ), maximum-likelihood (ML), and maximum-parsimony (MP) methods—were performed using MEGA 10 [ 8 ]. The reliability of each internal branch in the phylogenetic trees was tested using bootstrap analysis [ 9 ] with 1,000 random addition replicates. All sites with gaps were treated as having missing data. For the ML tree, the best substitution model was selected based on the Akaike information criterion in MEGA 10. RESULTS Identification of query 1 The reference strain with the highest similarity to query 1 was N. lacticolonia strain 001, with a value of 99.91% ( Fig 2a ; Link HAQ). This value exceeded the similarity value of the least similar combination within the N. lacticolonia clique group, CGMCC 3.18123 and URM8713, namely 99.44%; Fig 2a ; Link LAC). Thus, the queried strain was identified as being N. lacticolonia . Download figure Open in new tab Fig 1. Pipeline of ANI-netID. Download figure Open in new tab Fig 2. Results of identification based on ANI-netID. a) Query 1, b) Query 2, c) Query 3, d) Query 4, e) Query 5. Connected individuals have a higher value than the cut-off value. The dashed line connects the query to the individual with the highest similarity, indicating that the query does not belong to the same group as that individual. LAC: lowest ANI value in the clique-group, HAQ: highest ANI value to the query. Identification of query 2 Query 2 identification showed that this strain had the highest similarity (99.71 %) to N. singularis LC12068 ( Fig 2b ; Link HAQ). This value was less than the clique group threshold for N. singularis with 100% similarity between N. singularis LC12068 and CGMCC 3.19334 ( Fig 2b ; Link LAC). Therefore, query 2 does not belong to N. singularis but is closely related to N. singularis . Consequently, it is plausible that this lineage represents the most ancestral lineage of N. singularis . Alternatively, it may be a novel independent lineage. Consequently, the evaluation of query 2 by molecular phylogenetic and morphological analyses is necessary to determine its taxonomic position. Identification of query 3 The reference strain with the highest similarity to query 3 was N. lacticolonia 001 with a value of 100% ( Fig 2c ; link HAQ). Therefore, query 3 was identified as N. lacticolonia . Identification of query 4 Query 4 identification demonstrated that the reference strain to which this strain is most similar is N. sphaerica LC2839, with a value of 99.05% ( Fig 2d ; Link HAQ). This value is below the N. sphaerica clique-group threshold of 99.83% similarity between N. sphaerica LC2839 and LC2840 ( Fig 2d ; Link LAC). Consequently, query 4 does not belong to N. sphaerica but is a closely related lineage. Consequently, query 4 may represent either the most ancestral lineage of N. sphaerica or a new, independent lineage. Consequently, the evaluation of query 4 by molecular phylogenetic and morphological analyses is required. Identification of query 5 The reference strain with the highest homology to query 5 was N. chinensis CGMCC 3.18127, with a value of 100% ( Fig 2e ; Link HAQ). Thus, the query strain was identified to be N. chinensis . Evaluation of the identification result To evaluate the identification results from ANI-NetID, molecular phylogenetic analysis was performed on the dataset that had been used for identification and the five query strains, as well as the outgroups Apiospora malaysiana and A. pseudoparenchymatica , based on combined ITS, β-tubulin , and tef1 data ( Table 1 ). Following alignment, 1,148 sites were included, comprising 515 variable sites, 445 parsimony-informative sites, and 69 singleton sites. In the resulting ML, MP, and NJ phylogenetic trees, the strains of the same species (clique group) formed a monophyletic lineage ( Fig 3 ). Download figure Open in new tab Fig 3. Maximum-likelihood phylograms of Nigrospora species inferred from the concatenation of the ITS, β-tubulin , and tef1 genes. Values on the branches are ML, MP, and NJ bootstrap values (ML/MP/NJ). Strains with an asterisk are Ex-type strains. Queries are highlighted in bold. Queries 1 and 3 belonged to the N. lacticolonia clade ( Fig 3 ; bootstrap values BS, ML/MP/NJ: 97/99/100). Within this clade, both the strains formed a monophyletic group with N. lacticolonia 001 ( Fig 4a ). Query 2 was a sister to N. camelliae-sinensis and shared a common ancestor with N. singuralis ( Fig 3 ; ML/MP/NJ: 88/96/89; Fig 4b ). Query 4 was the ancestral lineage of the N. sphaerica clade ( Fig 3 ; ML/MP/NJ: 100/100/100; Fig 4c ). Query 5 belonged to the N. chinensis clade ( Fig 3 ; ML/MP/NJ: 99/100/100; Fig 4d ). These results support the ANI-netID results ( Fig 2 ). Download figure Open in new tab Fig 4. Parts of the maximum-likelihood cladograms (see Fig 3 ). N. lacticolonia clade (a), N. camelliae-sinensis and N. singularis clades (b), N. sphaerica clade (c), and N. chinensis clade (d). Strains with an asterisk are ex-type strains. Queries are highlighted in bold. The branch length is not related to evolutionary distance. Discussion This study introduces a new identification system called ANI-netID. This method differs from conventional identification methods in that it is based on a “clique-group” unit. This is a group of individuals of a species or genus based on similarity values between individuals. In this system, the similarity threshold for identification is automatically determined for each clique group. In other words, the similarity value between the two least similar individuals in the clique group is the threshold. This clarifies whether the individual to be identified belongs to the clique group. This addresses the issue of species identification failure caused by multiple high-scoring hits in BLAST searches [ 5 ]. In addition, the system uses average similarity when multiple DNA regions are used. This overcomes the problem of BLAST searches in individual regions for species identification, resulting in hits for different species in different regions. However, as with the DNA barcoding method for species identification, there is no universal DNA region common to all life forms [ 10 ]. Therefore, this system requires identification of the DNA regions used for each taxon. Therefore, the pre-determination of higher taxonomic ranks should be based on morphological analysis or the ITS region. In this study, an ANI-netID system was constructed for Nigrospora spp. and species identification was attempted for five strains based on their ITS, TUB and TEF regions. The identification results of this system were consistent with the results of the molecular phylogenetic analysis ( Figs 2 , 3 , and 4 ). The system uses the rate of matching nucleotide sites between two sequences as its basis. This has the same meaning as the p-distance, which calculates the rate of different nucleotide sites. The distances between individuals were similar to the phylogenetic relationships based on the p distance. Therefore, this system is compatible with the taxonomic systems established by molecular phylogenetic analysis. This likely produced results reflecting the taxonomic system. However, biases in nucleotide substitution rates due to codon positions [ 11 – 16 ], etc. are not considered. Therefore, the ANI-netID identification system cannot be used in cases where the grouping of taxa constructed by phylogenetic analysis based on complex mathematical models is different from that obtained by the p distance (e.g., when the same species is polyphyletic). Developers must ensure that the groupings for each target taxon match the groupings of the system before building the system. As a first step, it is recommended to check whether the phylogenetic tree based on the p distance is consistent with the current classification. The program developed in this study included a simple tool to check groupings (S1 dataset). As shown in Fig 1 S, it can be verified that populations that should belong to the same clique-group can be grouped without contamination of individuals from other clique-groups by exceeding the threshold. Although a simulation of species identification was conducted in this study using Nigrospora sp. as an example, the development of this identification system cannot be applied only to species identification. For example, three genera related to Pestalotiopsis ( Pestalotiopsis , Neopestalotiopsis and Pseudopestalotiopsis ) are similar in their branchial morphology [ 17 ], making them difficult to determine. In such cases, a system can be created to determine the genus based on the ANI-netIDs. In addition, agricultural and clinically important fungi such as the genera Colletotrichum and Fusarium are taxonomically studied intensively and have a large number of species. Species identification is conducted in molecular phylogenetic groups called species complexes, which are established under the genus [ 18 , 19 ]. Therefore, it is necessary to determine the species complexes before species identification. This system is expected to be effective in such cases. In addition to ANI-netID, an important aspect of species identification is the accumulation of longer sequences as data. The longer the sequence, the more information (mutations) that can be used to identify the individual and the more accurate the identification [ 20 ]. Recent developments in next-generation sequencing technologies have made it possible to obtain sequences that are longer than traditional barcode regions. Taxonomic studies have also begun to consider taxonomic systems based on phylogenetic relationships derived from genome-scale analyses [ 21 , 22 ]. In the future, as more genomic data accumulate, more taxa will be organized. This will increase the amount of data available for ANI-netID, and longer sequences will become available, leading to improved identification accuracy. We plan to make ANI-netID a platform where any taxonomist can develop a system for each taxonomic group individually and provide access to users who need identification work. This will allow more data to be accumulated and identified using ANI-netID for various taxonomic groups. Supporting Information S1 Fig. Networks formed by eliminating the threshold values. S1 Dataset . Programs and tested datasets. Author Contributions Conceptualization: Kyoko Watanabe, Shunsuke Nozawa. Formal analysis: Shunsuke Nozawa. Funding acquisition: Kyoko Watanabe. Investigation: Kyoko Watanabe, Shunsuke Nozawa. Methodology: Shunsuke Nozawa. Project administration: Kyoko Watanabe. Software: Shunsuke Nozawa. Supervision: Kyoko Watanabe. Validation: Kyoko Watanabe, Shunsuke Nozawa. Visualization: Shunsuke Nozawa. Writing – original draft: Shunsuke Nozawa. Writing – review & editing: Kyoko Watanabe. Acknowledgments We thank Dr. Yosuke Seto of the Cancer Chemotherapy Center, Japanese Foundation for Cancer Research, and Dr. Takayuki Aoki of National Agriculture and Food Research Organization for their invaluable advice. References 1. ↵ Kõljalg U , Larsson KH , Abarenkov K , Nilsson RH , Alexander IJ , Eberhardt U , et al. UNITE: a database providing web-based methods for the molecular identification of ectomycorrhizal fungi . New Phytol . 2005 ; 166 : 1063 – 1068 . doi: 10.1111/j.1469-8137.2005.01376.x . OpenUrl CrossRef PubMed Web of Science 2. ↵ White JR , Maddox C , White O , Angiuoli SV , Fricke WF , Clo VR-ITS . CloVR-ITS: Automated internal transcribed spacer amplicon sequence analysis pipeline for the characterization of fungal microbiota . Microbiome . 2013 ; 1 : 6 . doi: 10.1186/2049-2618-1-6 . OpenUrl CrossRef PubMed 3. Loos D , Zhang L , Beemelmanns C , Kurzai O , Panagiotou G . DAnIEL: A user-friendly web server for fungal ITS amplicon sequencing data . Front Microbiol . 2021 ; 12 : 720513 . doi: 10.3389/fmicb.2021.720513 . " OpenUrl CrossRef PubMed 4. ↵ Gweon HS , Oliver A , Taylor J , Booth T , Gibbs M , Read DS , et al. PIPITS: an automated pipeline for analyses of fungal internal transcribed spacer sequences from the Illumina sequencing platform . Methods Ecol Evol . 2015 ; 6 : 973 – 980 . doi: 10.1111/2041-210X.12399 . " OpenUrl CrossRef PubMed 5. ↵ Cai F , Druzhinina IS . In honor of John Bissett: authoritative guidelines on molecular identification of Trichoderma . Fungal Divers . 2021 ; 107 : 1 – 69 . doi: 10.1007/s13225-020-00464-4 . OpenUrl CrossRef 6. ↵ Steenwyk JL , Li Y , Zhou X , Shen XX , Rokas A . Incongruence in the phylogenomics era . Nat Rev Genet . 2023 ; 24 : 834 – 850 . doi: 10.1038/s41576-023-00620-x . OpenUrl CrossRef 7. ↵ Katoh K , Standley DM . MAFFT multiple sequence alignment software version 7: improvements in performance and usability . Mol Biol Evol . 2013 ; 30 : 772 – 780 . doi: 10.1093/molbev/mst010 . OpenUrl CrossRef PubMed Web of Science 8. ↵ Kumar S , Stecher G , Li M , Knyaz C , Tamura K , Mega X , et al. MEGA X: Molecular evolutionary genetics analysis across computing platforms . Mol Biol Evol . 2018 ; 35 : 1547 – 1549 . doi: 10.1093/molbev/msy096 . OpenUrl CrossRef PubMed 9. ↵ Felsenstein J . Confidence limits on phylogenies: an approach using the bootstrap . Evolution . 1985 ; 39 : 783 – 791 . doi: 10.1111/j.1558-5646.1985.tb00420.x . OpenUrl CrossRef GeoRef PubMed Web of Science 10. ↵ Ahmed S , Ibrahim M , Nantasenamat C , Nisar MF , Malik AA , Waheed R , et al. Pragmatic applications and universality of DNA barcoding for substantial organisms at species level: a review to explore a way forward . BioMed Res Int . 2022 ; 2022 : 1846485 . doi: 10.1155/2022/1846485 . OpenUrl CrossRef PubMed 11. ↵ Muto A , Osawa S . The guanine and cytosine content of genomic DNA and bacterial evolution . Proc Natl Acad Sci U S A . 1987 ; 84 : 166 – 169 . doi: 10.1073/pnas.84.1.166 . OpenUrl Abstract / FREE Full Text 12. Fitch WM . Evidence suggesting a non-random character to nucleotide replacements in naturally occurring mutations . J Mol Biol . 1967 ; 26 : 499 – 507 . doi: 10.1016/0022-2836(67)90317-8 . OpenUrl CrossRef PubMed Web of Science 13. Gojobori T , Li WH , Graur D . Patterns of nucleotide substitution in pseudogenes and functional genes . J Mol Evol . 1982 ; 18 : 360 – 369 . doi: 10.1007/BF01733904 . OpenUrl CrossRef PubMed Web of Science 14. Felsenstein J . Cases in which parsimony or compatibility methods will be positively misleading . Syst Zool . 1978 ; 27 : 401 – 410 . doi: 10.2307/2412923 . OpenUrl CrossRef 15. Sueoka N . On the genetic basis of variation and heterogeneity of DNA base composition . Proc Natl Acad Sci U S A . 1962 ; 48 : 582 – 592 . doi: 10.1073/pnas.48.4.582 . OpenUrl FREE Full Text 16. ↵ Freese E . On the evolution of the base composition of DNA . J Theor Biol . 1962 ; 3 : 82 – 101 . doi: 10.1016/S0022-5193(62)80005-8 . OpenUrl CrossRef Web of Science 17. ↵ Maharachchikumbura SSN , Hyde KD , Groenewald JZ , Xu J , Crous PW. Pestalotiopsis revisited . Stud Mycol . 2014 ; 79 : 121 – 186 . doi: 10.1016/j.simyco.2014.09.005 . OpenUrl CrossRef PubMed 18. ↵ da Silva LL , Moreno HLA , Correia HLN , Santana MF , de Queiroz MV. Colletotrichum : species complexes, lifestyle, and peculiarities of some sources of genetic variability . Appl Microbiol Biotechnol. AMB . 2020 ; 104 : 1891 – 1904 . doi: 10.1007/s00253-020-10363-y . OpenUrl CrossRef 19. ↵ Aoki T , O’Donnell K , Geiser DM. Systematics of key phytopathogenic Fusarium species: current status and future challenges . J Gen Plant Pathol . 2014 ; 80 : 189 – 201 . doi: 10.1007/s10327-014-0509-3 . OpenUrl CrossRef 20. ↵ Hillis DM , Huelsenbeck JP , Cunningham CW. Application and accuracy of molecular phylogenies . Science . 1994 ; 264 : 671 – 677 . doi: 10.1126/science.8171318 . OpenUrl Abstract / FREE Full Text 21. ↵ Seto Y , Iwasaki Y , Ogawa Y , Tamura K , Toda MJ. Skeleton phylogeny reconstructed with transcriptomes for the tribe Drosophilini (Diptera: Drosophilidae ) . Mol Phylogenet Evol . 2024 ; 191 : 107978 . doi: 10.1016/j.ympev.2023.107978 . OpenUrl CrossRef PubMed 22. ↵ Li Y , Steenwyk JL , Chang Y , Wang Y , James TY , Stajich JE , et al. A genome-scale phylogeny of the kingdom Fungi . Curr Biol . 2021 ; 31 : 1653 – 1665.e5 . doi: 10.1016/j.cub.2021.01.074 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted April 20, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following ANI-netID: a genome similarity-network based biological identification system Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share ANI-netID: a genome similarity-network based biological identification system S. Nozawa , K. Watanabe bioRxiv 2025.04.19.649685; doi: https://doi.org/10.1101/2025.04.19.649685 Share This Article: Copy Citation Tools ANI-netID: a genome similarity-network based biological identification system S. Nozawa , K. Watanabe bioRxiv 2025.04.19.649685; doi: https://doi.org/10.1101/2025.04.19.649685 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Microbiology Subject Areas All Articles Animal Behavior and Cognition (7622) Biochemistry (17650) Bioengineering (13871) Bioinformatics (41881) Biophysics (21424) Cancer Biology (18566) Cell Biology (25461) Clinical Trials (138) Developmental Biology (13365) Ecology (19866) Epidemiology (2067) Evolutionary Biology (24290) Genetics (15590) Genomics (22476) Immunology (17713) Microbiology (40331) Molecular Biology (17148) Neuroscience (88473) Paleontology (666) Pathology (2827) Pharmacology and Toxicology (4816) Physiology (7635) Plant Biology (15114) Scientific Communication and Education (2044) Synthetic Biology (4286) Systems Biology (9815) Zoology (2268)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-22T02:00:06.705733+00:00
License: CC-BY-4.0