PhyloControl: a phylogeny visualisation platform for risk analysis in weed biological control

preprint OA: closed CC-BY-NC-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Phylogenetic distance is a key measure used to develop species lists for host specificity tests that delimit the fundamental and realised host range of candidate weed biocontrol agents to meet the assumptions of the centrifugal phylogenetic method. Plant pathogens and insects, even those with broad host ranges, exhibit some degree of phylogenetic conservatism in their host plant associations. Thorough host-specificity testing is crucial to minimise the risk of off-target damage by biocontrol agents to native and economically important plant species. To facilitate this, host test lists need to be developed from an understanding of evolutionary relationships, usually visualised as a phylogenetic tree generated from genetic data, together with plant functional traits and geospatial information. Currently, the process of obtaining a host test list is not standardised, and the manual steps are time-consuming and challenging. We introduce a user-friendly visualisation tool called PhyloControl to aid researchers in their decision-making during biocontrol risk analysis. PhyloControl integrates taxonomic data, molecular data, spatial data, and plant traits in an intuitive interactive interface, empowering biocontrol practitioners to summarise, visualise and analyse data efficiently. Comprehensively sampled phylogenetic trees are often unavailable, and older published phylogenies often lack branch resolution and support, which increases uncertainty. PhyloControl includes a workflow implemented through Quarto notebooks in R that allows users to download publicly available DNA sequences and perform phylogenetic analyses. The modular workflow also incorporates species distribution modelling to predict the current and potential extent of target weed species. PhyloControl will streamline the development of biocontrol host tests lists to support risk analysis and decision making in classical weed biological control.
Full text 43,373 characters · extracted from preprint-html · click to expand
PhyloControl: a phylogeny visualisation platform for risk analysis in weed biological control | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results PhyloControl: a phylogeny visualisation platform for risk analysis in weed biological control View ORCID Profile Stephanie H. Chen , Lauren Stevens , View ORCID Profile Ben Gooden , View ORCID Profile Michelle A. Rafter , View ORCID Profile Nunzio Knerr , Peter H. Thrall , Louise Ord , View ORCID Profile Alexander N. Schmidt-Lebuhn doi: https://doi.org/10.1101/2025.06.11.658203 Stephanie H. Chen 1 CSIRO, Centre for Australian National Biodiversity Research (a joint venture between Parks Australia and CSIRO) , GPO Box 1700, Canberra ACT 2601, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Stephanie H. Chen Lauren Stevens 2 CSIRO Information Management and Technology , Clayton VIC 3168, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ben Gooden 3 CSIRO Health and Biosecurity , GPO Box 1700, Canberra ACT 2601, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ben Gooden Michelle A. Rafter 4 CSIRO Health and Biosecurity , GPO Box 2583, Brisbane QLD 4001, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michelle A. Rafter Nunzio Knerr 1 CSIRO, Centre for Australian National Biodiversity Research (a joint venture between Parks Australia and CSIRO) , GPO Box 1700, Canberra ACT 2601, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Nunzio Knerr Peter H. Thrall 5 CSIRO National Collections & Marine Infrastructure , GPO Box 1700, Canberra ACT 2601, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site Louise Ord 6 CSIRO Information Management and Technology , Eveleigh NSW 2015, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site Alexander N. Schmidt-Lebuhn 1 CSIRO, Centre for Australian National Biodiversity Research (a joint venture between Parks Australia and CSIRO) , GPO Box 1700, Canberra ACT 2601, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Alexander N. Schmidt-Lebuhn For correspondence: alexander.s-l{at}csiro.au Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Phylogenetic distance is a key measure used to develop species lists for host specificity tests that delimit the fundamental and realised host range of candidate weed biocontrol agents to meet the assumptions of the centrifugal phylogenetic method. Plant pathogens and insects, even those with broad host ranges, exhibit some degree of phylogenetic conservatism in their host plant associations. Thorough host-specificity testing is crucial to minimise the risk of off-target damage by biocontrol agents to native and economically important plant species. To facilitate this, host test lists need to be developed from an understanding of evolutionary relationships, usually visualised as a phylogenetic tree generated from genetic data, together with plant functional traits and geospatial information. Currently, the process of obtaining a host test list is not standardised, and the manual steps are time-consuming and challenging. We introduce a user-friendly visualisation tool called PhyloControl to aid researchers in their decision-making during biocontrol risk analysis. PhyloControl integrates taxonomic data, molecular data, spatial data, and plant traits in an intuitive interactive interface, empowering biocontrol practitioners to summarise, visualise and analyse data efficiently. Comprehensively sampled phylogenetic trees are often unavailable, and older published phylogenies often lack branch resolution and support, which increases uncertainty. PhyloControl includes a workflow implemented through Quarto notebooks in R that allows users to download publicly available DNA sequences and perform phylogenetic analyses. The modular workflow also incorporates species distribution modelling to predict the current and potential extent of target weed species. PhyloControl will streamline the development of biocontrol host tests lists to support risk analysis and decision making in classical weed biological control. Introduction In classical weed biological control, the release of a biocontrol agent is contingent on host-specificity testing to assess the risk of off-target damage in jurisdictions such as Australia, Canada, New Zealand, South Africa, and the United States of America. This risk assessment process seeks to ensure that agents are highly specific to the target weed. In accordance with the centrifugal phylogenetic method ( Wapshere, 1974 ) and its modernisations ( Briese, 2005 ; Sheppard et al., 2005 ), relatedness is an important factor in determining the host range of a candidate biocontrol agent. Closely related plants are more likely to have similar traits due to shared evolutionary history, and thus non-target plants closely related to the target weed are more likely to have similar or overlapping host recognition cues to candidate control agents ( Jones et al., 2020 , 2022 ; Le Falchier et al., 2025 ). In the absence of formal analyses, estimates of relatedness may be based on ranks in the taxonomic classification (same genus, same tribe, same subfamily, etc.). Since growing numbers of phylogenetic analyses have been published, relatedness has often been expressed in degrees of phylogenetic separation, as the count of inferred lineage splits separating a target weed from its common ancestor with another species in the same phylogenetic tree ( Kelch & McClay, 2004 ). An alternative metric of relatedness is patristic distance (also known as phylogenetic distance), which is the sum of branch lengths separating a target weed from another species across the phylogenetic tree. Patristic distance is preferred over degrees of separation as it can more accurately account for evolutionary relationships and variation in rates of evolution between species since the measure makes use of branch length information ( Chen et al., 2024 ). A host test list guides the specificity testing process and subsequent risk analysis as it helps define the scope of testing whilst prioritising species for testing ( Chen et al., 2024 ). However, currently there is no standard approach to creating a host test list in weed biocontrol control, which can often be a difficult and long process, resulting in bottlenecks to the release of an agent to manage target weeds. Sequencing data and species occurrence records are becoming increasingly available in databases such as GenBank and the Global Biodiversity Information Facility (GBIF) and may be leveraged for biocontrol. DNA sequences can be used to understand evolutionary relationships between species through phylogenetic analyses, while occurrence data may be used for species distribution modelling and to visualise spatial overlap between species. We introduce a modular workflow for generating inputs for visualisation via Quarto notebooks in R as well as an R Shiny application for visualisation. PhyloControl brings together genetic data in the form of a phylogenetic tree, spatial data, including species distribution modelling, and plant traits ( Figure 1 ). The relatedness of species is summarised through phylogenetic distance measures, namely degrees of separation and patristic (phylogenetic) distance. This user-friendly decision support tool is designed for both biocontrol practitioners and regulators. We anticipate that PhyloControl will make the process of developing a host test list more efficient as well as transparent and reproducible. Download figure Open in new tab Figure 1. PhyloControl Quarto notebooks and visualisation app inputs. The scientific name of the target weed and country that the biocontrol project will take place in are needed as initial inputs for the first Quarto notebook to generate the species list. Flow diagram created in Lucid (lucid.co). Generating inputs for PhyloControl A series of four Quarto notebooks has been developed to generate inputs for the visualisation app. The workflow is modular and reproducible which provides flexibility. For example, if a practitioner already has a species list or has generated data from their own sequencing experiment, these components can be skipped in the notebooks. Creating the species list (1_Species_list.qmd) The kewr R package ( Walker, 2024 ) is used to retrieve the scientific name of accepted species for a given plant family for specific location(s). The package uses Plants of the World Online, whose taxonomic backbone is the World Checklist of Vascular Plants. The output is a csv file of a list of species. Downloading and filtering global point observation data (2_Occurrences.qmd) Point observation data is retrieved from GBIF using the R package rgbif ( Chamberlain & Boettiger, 2017 ). We chose to use preserved specimens (i.e., herbarium specimens), which have coordinates attached and are also expertly identified and curated. The coords2country function then attaches a country to each record for use in the R Shiny application. Species distribution modelling with MaxEnt and CLIMATCH (3_Species_distribution_modelling.qmd) To predict the probability of occurrence for each species, we provide two methods for species distribution modelling: MaxEnt ( Phillips et al., 2006 ), implementing the Java source code ( Phillips et al., 2017 ), and CLIMATCH ( Erickson et al., 2022 ). MaxEnt (maximum entropy modelling) is a widely used correlative machine-learning algorithm used for estimating habitat suitability based on presence-only data and environmental predictors. It is often chosen over other methods due to its predictive accuracy and ease of use ( Merow et al., 2013 ). Meanwhile, CLIMATCH is a climate-matching tool that compares known species locations to locations to each possible invasion location and searches for the best match in a one-to-one association based on environmental variables, providing a conservative perspective compared to models that look for many-to-many associations. It requires the same presence-only data and environmental layers as input. It is a simpler approach compared to other SDM methods but is quick to run; government agencies in Australia and the United States use CLIMATCH as validation studies show that it is consistent predictor of invasion risk ( Bomford et al., 2009 ; Hayes & Barry, 2008 ). The parameterisation of CLIMATCH intentionally resembles the CLIMATCH v2.0 webtool developed by the Australian Bureau of Agricultural and Resource Economics and Sciences ( ABARES, 2020 ). The climate layers used are the CHELSA v2.1 climatologies ( Karger et al., 2017 , 2021 ). Specifically, the variables starting with ‘bio’ that are not derived from other variables (i.e., bio1-19, excluding bio2-4) are accessed via the R package ClimDatDownloadR ( Jentsch et al., 2023 ). These variables include mean annual air temperature (bio1), mean daily maximum air temperature of the warmest month (bio5), mean daily minimum air temperature of the coldest month (bio6), annual range of air temperature (bio7), mean daily mean air temperatures of the wettest quarter (bio8), mean daily mean air temperatures of the driest quarter (bio9), mean daily mean air temperatures of the warmest quarter (bio10), mean daily mean air temperatures of the coldest quarter (bio11), annual precipitation amount (bio12), precipitation amount of the wettest month (bio13), precipitation amount of the driest month (bio14), precipitation seasonality (bio15), mean monthly precipitation amount of the wettest quarter (bio16), mean monthly precipitation amount of the driest quarter (bio17), mean monthly precipitation amount of the warmest quarter (bio18), and mean monthly precipitation amount of the coldest quarter (bio19). CLIMATCH is set to run for the extent of Australia by default, but may be used for other countries while MaxEnt is run at the global extent by default. For MaxEnt, the complementary log-log (cloglog) link function ( Phillips et al., 2017 ) was used. Four threshold values were selected for visualisation – the maximum of Cohen’s Kappa ( Monserud & Leemans, 1992 ), the maximum training sensitivity plus specificity ( Cantor et al., 1999 ), threshold at which the sensitivity and specificity are equal, and sensitivity at 0.9. These were ordered and used as the boundary values for five bins. The top two bins were coloured red on the map, the next was orange, and the next was yellow. The bottom bin remained transparent. CLIMATCH outputs a score between 0 to 10, indicating the degree of the match and results were visualised in a rainbow gradient palette. Building a phylogeny with public sequence data from GenBank 4_Sequences_and_phylogenomics.qmd) The R package phruta ( Román-Palacios, 2023 ) is used to find and download the markers available in GenBank and align the DNA sequences. The markers chosen can be set by a threshold where the default is to use markers that are found in at least 30 % of taxa in the species list provided. The IPS package is used for phylogenetic analysis with RAxML ( Stamatakis, 2006 ). This includes a rapid bootstrap analysis and search for the best-scoring maximum likelihood tree as a quick way of generating a tree that is compatible with a local computer. Users can also generate their own sequences and phylogenies using alternative workflows (e.g. Chen et al., 2024 ), and the resulting Newick-formatted tree file may be used as input for visualisation in the PhyloControl R Shiny application. Example use case with comparison of different starting species lists The code contains an example for a biocontrol project on the target weed flaxleaf fleabane, ( Erigeron bonariensis ) ( CSIRO, 2023 ). We also compared three different scenarios where different levels of sequencing data and prior knowledge about the ingroup were available. For Scenario 1, we generated a species list containing all Asteraceae occurring in Australia. In total, this species list had 1,425 species. An outgroup from the sister family Calyceraceae, Boopis anthemoides , was added for rooting. Of those species, 563 were not found and returned errors when GenBank’s taxonomic backbone was queried. There were three markers available for at least 30% of the 863 taxa in the cleaned species vector – matK (57%), ITS1 (32%), and ITS2 (30%), resulting in a concatenated dataset of sequence data for 328 species. The overall execution time for the full maximum likelihood analysis (100 bootstraps) was 5.1 h with 8 threads on a MacBook Pro with an M2 chip. For Scenario 2, we focused on the tribe level and started with the species list from a target sequencing experiment that was curated by an Asteraceae botanist (i.e. from the experiment described for Scenario 3). Out of its 286 species, 102 gave errors when queries against GenBank’s taxonomic backbone, leaving 184 species to be input into the gene.sampling.retrieve function from the phruta package. There were 3 markers sampled in at least 30% of taxa: ITS1 (87%), ITS2 (85%), 5.8S rRNA (81%), resulting in a concatenated dataset covering 132 species. The overall execution time for the full maximum likelihood analysis was 0.36 h. For Scenario 3, we aimed to generate a species tree at the tribe level of the target weed, i.e., Asteraceae tribe Astereae, using the Angiosperms353 target capture bait kit which amplifies 353 nuclear genes. In this paper, we focus on the concatenated analysis, rather than the coalescent analyses. This approach of using target capture sequencing allowed for comprehensive sampling and high phylogenetic resolution. The final tree contained 279 taxa (some samples failed to sequence or were quality filtered out), including 4 outgroups from other tribes of Asteraceae. These analyses required High Performance Computing (HPC), and the resources required were detailed in the paper where the phylogeny was initially published ( Chen et al., 2024 ). PhyloControl visualisation tool The visualisation component of PhyloControl is an R Shiny application ( Figure 2 ). The required inputs for visualisation are the phylogenetic tree with optional components being the occurrence data, species distribution models, and plant traits ( Table 1 ). Instructional text is provided through the ‘Help’ button via rintrojs ( Ganz, 2016 ). Download figure Open in new tab Figure 2. The home landing tab of the PhyloControl R Shiny application. View this table: View inline View popup Table 1. Inputs for visualisation with the PhyloControl R Shiny application. The data displayed in the demo version of PhyloControl corresponds to Scenario 3 described above and is a previously published tribe level Astereae phylogeny using target sequence capture data ( Chen et al., 2024 ). For the GBIF records, the initial query for preserved specimens with coordinate information contained 4,170,267 occurrences ( GBIF.org , 2024). Following attaching country names and filtering, 85,679 records remained. For the species distribution modelling, the demo version of the app contains only the MaxEnt and CLIMATCH results for the target species. For demo data available on the DAP Portal, this is expanded to the six Erigeron species in the phylogeny. The modelling function took approximately a minute for each species. MaxEnt was run at a global scale while CLIMATCH was run only for Australia. A csv file with the native vs introduced status for each species in Australia was prepared manually. Phylogeny tab The APE ( Paradis et al., 2004 ) package was used for handling and manipulating the tree data while ggplot2 ( Wickham, 2009 ), plotly ( Sievert, 2020 ), and ggtree ( Yu et al., 2017 ) were used for visualisation. There are options provided for the phylogeny for re-rooting, collapsing and expanding sections of the tree, and converting it to a cladogram. The phylogeny should be outgroup rooted using the most evolutionarily distant species for correct downstream inferences. Additionally, branch support values can be visualised as shades of grey. Where multiple support values are found in a Newick file, the visualisation will show the first set of values. Up to four heatmaps as vertical bars may be simultaneously displayed to the right of the phylogenetic tree ( Figure 3 ). Categorical and/or numerical traits provided via a csv file as well as the phylogenetic distance measures degrees of separation and patristic distance, which are calculated in app, may be shown. For categorical data, the discrete colour scale is limited to five categories to facilitate visualisation. The scico package ( Crameri, 2018 ) was used for continuous scales. There is an option to highlight trait data in two colours: one colour for the traits with the same value as the target species and grey for all others. The visualisation may be downloaded as an image for use in reports, proposals, or manuscripts. Download figure Open in new tab Figure 3. Phylogeny of the tribe Astereae for a biological control project on the target weed Erigeron bonariensis . Heatmaps display native vs introduced status in Australia, degrees of separation, and patristic distance. Note the clades that include all Celmisia (39 species), Brachyscome (65 species), and Olearia (93 species) were collapsed to shorten the length of the image. Map tab Two species at a time can be selected to display occurrence records to assist in the visualisation of spatial overlap ( Figure 4 ). A multiselect menu is available to filter the records to selected countries. Download figure Open in new tab Figure 4. Global occurrence records from GBIF for the target weed flaxleaf fleabane ( Erigeron bonariensis ) in teal and the Australian native imbricate daisy bush ( Olearia imbricata ) in purple: A) All countries selected, B) Only Australia selected and C) Only occurrences in Australia displayed and zoomed to south-west Western Australia where the native daisy is endemic. Model tab The results of species distribution modelling are displayed on a map coloured by MaxEnt presence of probability or CLIMATCH score, depending on the algorithm(s) chosen in the Quarto notebooks ( Figure 5 ). Download figure Open in new tab Figure 5. Species distribution model using A) MaxEnt and B) CLIMATCH for Erigeron bonariensis cropped to Australia. Report tab A table of species in the tree ranked by the phylogenetic distance measures degrees of separation and patristic distance is available to download as a csv to assist a biocontrol practitioner to create a host test list. Discussion PhyloControl is applicable to biological control projects globally, providing a reproducible and transparent means to bring together and visualise genetic, spatial, and trait data to optimise risk analysis. By focusing on species most closely related to the target weed, PhyloControl enables more targeted testing, reducing unnecessary experiments while allowing for thorough testing downstream in the biocontrol pipeline. Typically, a practitioner would be able to run the entire workflow and visualisation app on a local computer within a day, reducing the time and effort typically required to produce a host test list. This not only streamlines regulatory approval processes but also strengthens the scientific basis for biocontrol decisions which ultimately supports safer and more effective weed management strategies. Gathering large volumes of data from disparate sources remains a challenge. Obtaining integrated taxonomic information is difficult and is a barrier to interoperability ( Sandall et al., 2023 ). For example, the taxonomic backbones of the databases used for the Quarto notebooks, World Checklist of Vascular Plants, GBIF, and GenBank, do not align completely. Nonetheless, the R packages wrap their respective APIs to query and retrieve data facilitates the PhyloControl workflow. For the final trees for Scenario 1 and Scenario 2 of the Erigeron data, Conyza , a synonym of Erigeron , appeared in the tree, so the final trees may still require some manual curation if the study group has a recent history of taxonomic change. A caveat of using publicly available sequencing data is that rare native species will likely be unsequenced, and these are some of the species important for biocontrol risk analysis where off-target damage to natives is a concern. As more genomics resources are generated, more comprehensive phylogenies will be able to be generated from GenBank data. In practice, biocontrol researchers will include threatened native relatives in the test list even if molecular data to understand evolutionary relationships are lacking. Here, PhyloControl allows for the identification of uncertain phylogenetic context so that important species may be targeted for additional sequencing. Data availability also varies greatly across plant families. For example, families such as Fabaceae have had significant sequencing efforts directed towards them due to their economic importance ( Group et al., 2013 ), so using GenBank data will result in a relatively comprehensive phylogeny. Checking for available GenBank data using the workflow may also help in the planning of a sequencing experiment to fill gaps and is a cost-effective way to obtain a comprehensive phylogeny. In general, genomics is being increasingly used in biological control in host testing lists and applications such as improving effectiveness of agents by targeting weed traits ( Leung et al., 2020 ). Likewise, collecting trait data from open sources across the internet remains a challenge. Databases such as TRY ( Kattge et al., 2020 ) and AusTraits ( Falster et al., 2021 ) exist, but information will likely be patchy for a study group that a practitioner is interested in, and often they do not have global coverage, so this step remains manual in this initial version of PhyloControl. Currently, the user needs to identify informative traits that underpin agent-plant interactions and populated a spreadsheet with information with information from the literature. Possible extensions of PhyloControl may include running the species distribution models with different climate change scenarios to understand effect of climate on invasion dynamics ( Booth, 2018 ). Accessibility may be enhanced through the visualisation app being hosted through a server online if there is demand from practitioners. Additionally, whilst the tool is currently tailored to weed biological control, it may be modified to source information from other databases that are focused on the biological control of organisms other than plants. PhyloControl addresses the challenges of a process that is currently unstandardised and time consuming. By integrating taxonomic, molecular, spatial, and trait data on target weeds and related plant species, PhyloControl empowers biocontrol practitioners to develop host test lists more efficiently, increasing confidence in risk analysis. Additionally, this improved approach has the potential to streamline regulatory by enhancing transparency and reproducibility. Therefore, we anticipate that the adoption of PhyloControl will lead to more robust and defensible risk assessments in weed biological control. Data availability The open source code for the PhyloControl R Shiny application is available at GitHub ( https://github.com/csiro/phylocontrol-viz ). The Quarto notebooks for generating the visualisation inputs are available at GitHub ( https://github.com/csiro/phylocontrol-geninput ). The outputs generated using the Quarto notebooks including the Erigeron demo dataset included with the app are available on the CSIRO Data Access Portal ( https://doi.org/10.25919/21fr-hk78 ). A demo version of the app is available at https://shiny.csiro.au/phylocontrol-viz-demo/ . Author contributions Alexander Schmidt-Lebuhn, Ben Gooden, and Michelle Rafter conceptualised the project. Stephanie Chen and Nunzio Knerr prepared and tested the Quarto notebooks. The base framework concept and code were originally developed by Louise Ord with conversion to bslib done by Lauren Stevens. The core PhyloControl application code (Phylogeny and Map tabs) was written and implemented by Lauren Stevens based on the approach and design conceptualised by Louise Ord. Advice on solutions and implementation strategies for technical challenges were provided by Louise Ord and Lauren Stevens. The Model tab was created and developed by Nunzio Knerr and Stephanie Chen. The Report tab was created by Stephanie Chen. Alexander Schmidt-Lebuhn wrote the original functions for calculating degrees of separation and patristic distance. The Erigeron datasets used in this paper were prepared by Stephanie Chen, Nunzio Knerr, and Alexander Schmidt-Lebuhn. Stephanie Chen led the writing of the manuscript. All authors contributed critically to the drafts and gave final approval for publication. Conflict of Interest statement The authors declare no conflict of interest. Acknowledgements We thank the attendees of the Annual Biological Control Workshop on 18 October 2023 at the Australian Government Department of Agriculture, Fisheries and Forestry for feedback on a prototype. The development of the R Shiny application was supported by the CSIRO Scientific Computing Collaboration Project programme. Footnotes https://shiny.csiro.au/phylocontrol-viz-demo/ References ↵ ABARES . ( 2020 ). Climatch (Version 2.0) [Computer software] . https://climatch.cp1.agriculture.gov.au/ ↵ Bomford , M. , Kraus , F. , Barry , S. C. , & Lawrence , E . ( 2009 ). Predicting establishment success for alien reptiles and amphibians: A role for climate matching . Biological Invasions , 11 ( 3 ), 713 – 724 . doi: 10.1007/s10530-008-9285-3 OpenUrl CrossRef ↵ Booth , T. H . ( 2018 ). Species distribution modelling tools and databases to assist managing forests under climate change . Forest Ecology and Management , 430 , 196 – 203 . doi: 10.1016/j.foreco.2018.08.019 OpenUrl CrossRef ↵ Briese , D. T . ( 2005 ). Translating host-specificity test results into the real world: The need to harmonize the yin and yang of current testing procedures . Biological Control , 35 ( 3 ), 208 – 214 . doi: 10.1016/j.biocontrol.2005.02.001 OpenUrl CrossRef Web of Science ↵ Cantor , S. B. , Sun , C. C. , Tortolero-Luna , G. , Richards-Kortum , R. , & Follen , M . ( 1999 ). A comparison of C/B ratios from studies using receiver operating characteristic curve analysis . Journal of Clinical Epidemiology , 52 ( 9 ), 885 – 892 . OpenUrl CrossRef PubMed Web of Science ↵ Chamberlain , S. A. , & Boettiger , C . ( 2017 ). R Python, and Ruby clients for GBIF species occurrence data (No. e3304v1) . PeerJ Inc . doi: 10.7287/peerj.preprints.3304v1 OpenUrl CrossRef ↵ Chen , S. H. , Gooden , B. , Rafter , M. A. , Hunter , G. C. , Grealy , A. , Knerr , N. , & Schmidt-Lebuhn , A. N . ( 2024 ). Phylogenomics-driven host test list selection for weed biological control . Biological Control , 193 , 105529 . doi: 10.1016/j.biocontrol.2024.105529 OpenUrl CrossRef ↵ Crameri , F . ( 2018 ). Scientific colour maps (Version 3.0.1) [Computer software] . Zenodo . doi: 10.5281/ZENODO.1243909 OpenUrl CrossRef ↵ CSIRO . ( 2023 ). Flaxleaf fleabane biological control: Previous research (2016-2023) . Flaxleaf Fleabane Biological Control . https://research.csiro.au/flaxleaf-fleabane/achievements-rrnd4p-rnd2/ ↵ Erickson , R. A. , Engelstad , P. S. , Jarnevich , C. S. , Sofaer , H. R. , & Daniel , W. M . ( 2022 ). Climate matching with the climatchR R package . Environmental Modelling & Software , 157 , 105510 . doi: 10.1016/j.envsoft.2022.105510 OpenUrl CrossRef ↵ Falster , D. , Gallagher , R. , Wenk , E. H. , Wright , I. J. , Indiarto , D. , Andrew , S. C. , Baxter , C. , Lawson , J. , Allen , S. , Fuchs , A. , Monro , A. , Kar , F. , Adams , M. A. , Ahrens , C. W. , Alfonzetti , M. , Angevin , T. , Apgaua , D. M. G. , Arndt , S. , Atkin , O. K. , … Ziemińska , K . ( 2021 ). AusTraits, a curated plant trait database for the Australian flora . Scientific Data , 8 ( 1 ), 254 . doi: 10.1038/s41597-021-01006-6 OpenUrl CrossRef ↵ Ganz , C . ( 2016 ). rintrojs: A Wrapper for the Intro.js Library . The Journal of Open Source Software , 1 ( 6 ), 63 . doi: 10.21105/joss.00063 OpenUrl CrossRef GBIF.org . ( 2024 ). Occurrence Download [Dataset] . The Global Biodiversity Information Facility . doi: 10.15468/DL.FGRZP9 OpenUrl CrossRef ↵ Group , T. L. P. W. , Bruneau , A. , Doyle , J. J. , Herendeen , P. , Hughes , C. , Kenicer , G. , Lewis , G. , Mackinder , B. , Pennington , R. T. , Sanderson , M. J. , Wojciechowski , M. F. , Boatwright , S. , Brown , G. , Cardoso , D. , Crisp , M. , Egan , A. , Fortunato , R. H. , Hawkins , J. , Kajitap , T. , … van Wyk , B.-E. ( 2013 ). Legume phylogeny and classification in the 21st century: Progress, prospects and lessons for other species–rich clades . TAXON , 62 ( 2 ), 217 – 248 . doi: 10.12705/622.8 OpenUrl CrossRef Web of Science ↵ Hayes , K. R. , & Barry , S. C . ( 2008 ). Are there any consistent predictors of invasion success? Biological Invasions , 10 ( 4 ), 483 – 506 . doi: 10.1007/s10530-007-9146-5 OpenUrl CrossRef Web of Science ↵ Jentsch , H. , Weidinger , J. , & Bobrowski , M . ( 2023 ). ClimDatDownloadR: Downloads Climate Data from Chelsa and WorldClim (Version 0.1.7) [Computer software] . Zenodo . doi: 10.5281/zenodo.7924343 OpenUrl CrossRef ↵ Jones , L. C. , Rafter , M. A. , & Walter , G. H . ( 2020 ). Host plant acceptance in a generalist insect: Threshold, feedback or choice? Behaviour , 157 ( 12–13 ), 1059 – 1089 . doi: 10.1163/1568539X-bja10041 OpenUrl CrossRef ↵ Jones , L. C. , Rafter , M. A. , & Walter , G. H . ( 2022 ). Host interaction mechanisms in herbivorous insects – life cycles, host specialization and speciation . Biological Journal of the Linnean Society , 137 ( 1 ), 1 – 14 . doi: 10.1093/biolinnean/blac070 OpenUrl CrossRef ↵ Karger , D. N. , Conrad , O. , Böhner , J. , Kawohl , T. , Kreft , H. , Soria-Auza , R. W. , Zimmermann , N. E. , Linder , H. P. , & Kessler , M . ( 2017 ). Climatologies at high resolution for the earth’s land surface areas . Scientific Data , 4 ( 1 ), 170122 . doi: 10.1038/sdata.2017.122 OpenUrl CrossRef PubMed ↵ Karger , D. N. , Conrad , O. , Böhner , J. , Kawohl , T. , Kreft , H. , Soria-Auza , R. W. , Zimmermann , N. E. , Linder , H. P. , & Kessler , M . ( 2021 ). Climatologies at high resolution for the earth’s land surface areas [Dataset] . EnviDat . doi: 10.16904/envidat.228 OpenUrl CrossRef ↵ Kattge , J. , Bönisch , G. , Díaz , S. , Lavorel , S. , Prentice , I. C. , Leadley , P. , Tautenhahn , S. , Werner , G. D. A. , Aakala , T. , Abedi , M. , Acosta , A. T. R. , Adamidis , G. C. , Adamson , K. , Aiba , M. , Albert , C. H. , Alcántara , J. M. , Alcázar C , C. , Aleixo , I. , Ali , H. , … Wirth , C . ( 2020 ). TRY plant trait database – enhanced coverage and open access . Global Change Biology , 26 ( 1 ), 119 – 188 . doi: 10.1111/gcb.14904 OpenUrl CrossRef ↵ Kelch , D. G. , & McClay , A . ( 2004 ). Putting the phylogeny into the centrifugal phylogenetic method . Proceedings of the XI International Symposium on Biological Control of Weeds. XI International Symposium on Biological Control of Weeds . ↵ Le Falchier , E. J. , Telmadarrehei , T. , Rafter , M. A. , & Minteer , C. R. ( 2025 ). One size does not fit all: Classical weed biological control across continents . Biological Control , 200 , 105661 . doi: 10.1016/j.biocontrol.2024.105661 OpenUrl CrossRef ↵ Leung , K. , Ras , E. , Ferguson , K. B. , Ariëns , S. , Babendreier , D. , Bijma , P. , Bourtzis , K. , Brodeur , J. , Bruins , M. A. , Centurión , A. , Chattington , S. R. , C hinchilla-Ramírez , M. , Dicke , M. , Fatouros , N. E. , González-Cabrera , J. , Groot , T. V. M. , Haye , T. , Knapp , M. , Koskinioti , P. , … Pannebakker , B. A. ( 2020 ). Next-generation biological control: The need for integrating genetics and genomics . Biological Reviews of the Cambridge Philosophical Society , 95 ( 6 ), 1838 – 1854 . doi: 10.1111/brv.12641 OpenUrl CrossRef ↵ Merow , C. , Smith , M. J. , & Silander Jr , J. A . ( 2013 ). A practical guide to MaxEnt for modeling species’ distributions: What it does, and why inputs and settings matter . Ecography , 36 ( 10 ), 1058 – 1069 . doi: 10.1111/j.1600-0587.2013.07872.x OpenUrl CrossRef PubMed Web of Science ↵ Monserud , R. A. , & Leemans , R . ( 1992 ). Comparing global vegetation maps with the Kappa statistic . Ecological Modelling , 62 ( 4 ), 275 – 293 . OpenUrl CrossRef Web of Science ↵ Paradis , E. , Claude , J. , & Strimmer , K . ( 2004 ). APE: Analyses of Phylogenetics and Evolution in R language . Bioinformatics , 20 ( 2 ), 289 – 290 . doi: 10.1093/bioinformatics/btg412 OpenUrl CrossRef PubMed Web of Science ↵ Phillips , S. J. , Anderson , R. P. , Dudík , M. , Schapire , R. E. , & Blair , M. E . ( 2017 ). Opening the black box: An open-source release of Maxent . Ecography , 40 ( 7 ), 887 – 893 . doi: 10.1111/ecog.03049 OpenUrl CrossRef PubMed ↵ Phillips , S. J. , Anderson , R. P. , & Schapire , R. E . ( 2006 ). Maximum entropy modeling of species geographic distributions . Ecological Modelling , 190 ( 3 ), 231 – 259 . doi: 10.1016/j.ecolmodel.2005.03.026 OpenUrl CrossRef PubMed ↵ Román-Palacios , C . ( 2023 ). The phruta r package: Increasing access, reproducibility and transparency in phylogenetic analyses . Methods in Ecology and Evolution , 14 ( 9 ), 2284 – 2299 . doi: 10.1111/2041-210X.14147 OpenUrl CrossRef ↵ Sandall , E. L. , Maureaud , A. A. , Guralnick , R. , McGeoch , M. A. , Sica , Y. V. , Rogan , M. S. , Booher , D. B. , Edwards , R. , Franz , N. , Ingenloff , K. , Lucas , M. , Marsh , C. J. , McGowan , J. , Pinkert , S. , Ranipeta , A. , Uetz , P. , Wieczorek , J. , & Jetz , W. ( 2023 ). A globally integrated structure of taxonomy to support biodiversity science and conservation . Trends in Ecology & Evolution , 38 ( 12 ), 1143 – 1153 . doi: 10.1016/j.tree.2023.08.004 OpenUrl CrossRef PubMed ↵ Sheppard , A. W. , van Klinken , R. D. , & Heard , T. A. ( 2005 ). Scientific advances in the analysis of direct risks of weed biological control agents to nontarget plants . Biological Control , 35 ( 3 ), 215 – 226 . doi: 10.1016/j.biocontrol.2005.05.010 OpenUrl CrossRef Web of Science ↵ Sievert , C . ( 2020 ). Interactive Web-Based Data Visualization with R, plotly, and shiny. CRC Press. https://books.google.com.au/books?id=7zPNDwAAQBAJ ↵ Stamatakis , A . ( 2006 ). RAxML-VI-HPC: maximum likelihood-based phylogenetic analyses with thousands of taxa and mixed models . Bioinformatics , 22 ( 21 ), 2688 – 2690 . OpenUrl CrossRef PubMed Web of Science ↵ Walker , B. ( 2024 ). kewr: R Package to Access Kew Data APIs (Version 0.6.1) [Computer software] . https://github.com/barnabywalker/kewr ↵ Wapshere , A. J . ( 1974 ). A strategy for evaluating the safety of organisms for biological weed control . Annals of Applied Biology , 77 ( 2 ), 201 – 211 . doi: 10.1111/j.1744-7348.1974.tb06886.x OpenUrl CrossRef Web of Science ↵ Wickham , H. ( 2009 ). ggplot2: Elegant Graphics for Data Analysis . Springer New York . https://books.google.com.au/books?id=bes-AAAAQBAJ ↵ Yu , G. , Smith , D. K. , Zhu , H. , Guan , Y. , & Lam , T. T.-Y . ( 2017 ). ggtree: An r package for visualization and annotation of phylogenetic trees with their covariates and other associated data . Methods in Ecology and Evolution , 8 ( 1 ), 28 – 36 . doi: 10.1111/2041-210X.12628 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted June 15, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following PhyloControl: a phylogeny visualisation platform for risk analysis in weed biological control Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share PhyloControl: a phylogeny visualisation platform for risk analysis in weed biological control Stephanie H. Chen , Lauren Stevens , Ben Gooden , Michelle A. Rafter , Nunzio Knerr , Peter H. Thrall , Louise Ord , Alexander N. Schmidt-Lebuhn bioRxiv 2025.06.11.658203; doi: https://doi.org/10.1101/2025.06.11.658203 Share This Article: Copy Citation Tools PhyloControl: a phylogeny visualisation platform for risk analysis in weed biological control Stephanie H. Chen , Lauren Stevens , Ben Gooden , Michelle A. Rafter , Nunzio Knerr , Peter H. Thrall , Louise Ord , Alexander N. Schmidt-Lebuhn bioRxiv 2025.06.11.658203; doi: https://doi.org/10.1101/2025.06.11.658203 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Plant Biology Subject Areas All Articles Animal Behavior and Cognition (7643) Biochemistry (17717) Bioengineering (13910) Bioinformatics (42017) Biophysics (21480) Cancer Biology (18628) Cell Biology (25537) Clinical Trials (138) Developmental Biology (13392) Ecology (19935) Epidemiology (2067) Evolutionary Biology (24356) Genetics (15617) Genomics (22530) Immunology (17755) Microbiology (40438) Molecular Biology (17200) Neuroscience (88705) Paleontology (667) Pathology (2840) Pharmacology and Toxicology (4832) Physiology (7657) Plant Biology (15171) Scientific Communication and Education (2046) Synthetic Biology (4304) Systems Biology (9828) Zoology (2272)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-22T02:00:06.705733+00:00
License: CC-BY-NC-4.0