Interpretable Machine Learning and Comparative Genomics Reveal Microbial Plastic-Degrading (Microbeyt) Potential

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Plastic pollution poses a critical environmental threat, and microbial enzymes represent a sustainable strategy for polymer degradation. We present a computational pipeline that integrates orthogroup-based genomic analysis with machine learning and interpretable feature importance to identify microbial strains with high plastic-degrading potential. Using presence or absence matrices and SHAP-derived feature contributions to the MTP visualization, the workflow highlights conserved gene modules driving predictive classification. Application to a single genus revealed strains harboring versatile enzymatic repertoires capable of targeting diverse polymers, including polyethylene, polyethylene terephthalate, polyurethane, and polyhydroxyalkanoates. These findings provide a rational framework for prioritizing candidate strains for experimental validation and bioremediation strategies. Overall, this study demonstrates how integrating comparative genomics with interpretable machine learning can guide the systematic discovery of microbial solutions to plastic pollution.
Full text 50,685 characters · extracted from preprint-html · click to expand
Interpretable Machine Learning and Comparative Genomics Reveal Microbial Plastic-Degrading (Microbeyt) Potential | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Interpretable Machine Learning and Comparative Genomics Reveal Microbial Plastic-Degrading (Microbeyt) Potential Lokendra S. Thakur , Gurpreet Bharj , Manish Saroya doi: https://doi.org/10.1101/2025.09.17.676701 Lokendra S. Thakur 1 Innovation Engine Lab (iEL), MdtRI , Hyderabad, INDIA 2 Department of Neurology, Massachusetts General Hospital, Harvard Medical School , Massachusetts, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: lthakur{at}mdtri.org Gurpreet Bharj 1 Innovation Engine Lab (iEL), MdtRI , Hyderabad, INDIA 3 Institute for Immunity, Transplantation, and Infection, Stanford University , Stanford, California, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Manish Saroya 1 Innovation Engine Lab (iEL), MdtRI , Hyderabad, INDIA 4 Honda Research Institute , Mountain View, California, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Plastic pollution poses a critical environmental threat, and microbial enzymes represent a sustainable strategy for polymer degradation. We present a computational pipeline that integrates orthogroup-based genomic analysis with machine learning and interpretable feature importance to identify microbial strains with high plastic-degrading potential. Using presence or absence matrices and SHAP-derived feature contributions to the MTP visualization, the workflow highlights conserved gene modules driving predictive classification. Application to a single genus revealed strains harboring versatile enzymatic repertoires capable of targeting diverse polymers, including polyethylene, polyethylene terephthalate, polyurethane, and polyhydroxyalkanoates. These findings provide a rational framework for prioritizing candidate strains for experimental validation and bioremediation strategies. Overall, this study demonstrates how integrating comparative genomics with interpretable machine learning can guide the systematic discovery of microbial solutions to plastic pollution. 1 Introduction Plastic waste has become one of the most persistent environmental contaminants due to its durability and resistance to biodegradation. The world produces approximately 350 million tonnes of plastic waste annually, with an estimated 19 to 23 million tonnes leaking into aquatic ecosystems each year. 1 This alarming trend contributes to the accumulation of 75 to 199 million tonnes of plastic waste currently in our oceans. 2 Despite efforts to mitigate this issue, only about 9.5% of the 400 million tonnes of plastic produced in 2022 was made from recycled materials, highlighting a significant reliance on fossil fuels for plastic production. 3 Furthermore, approximately 70% of plastic waste remains uncollected, leaks into the environment, is dumped into landfills, or subjected to open burning. 4 While several bacterial species harbor enzymes capable of attacking synthetic polymers, the genomic determinants underlying these capabilities remain poorly characterized. Recent studies have begun to address this gap. Metagenomic analysis of plastic-degrading microbial communities has revealed novel microorganisms, metabolic pathways, and biocatalysts involved in polymer degradation. 5 Additionally, genomic mining has identified novel polyethylene terephthalate (PET)-degrading enzymes, although challenges remain in enhancing their efficiency for practical applications. 6 The global plastics supply chain is highly complex, with trade-linked material flows amplifying environmental risks. 7 Microbial biodegradation has emerged as a promising avenue to mitigate these risks, converting plastic waste into valuable biochemical resources. 8 This study aims to systematically analyze genomic conservation patterns within a single genus to identify candidate strains with high plastic-degrading potential. By leveraging advanced genomic tools and understanding the metabolic pathways involved, we can develop more efficient bioremediation strategies to combat the escalating plastic pollution crisis. Our approach was inspired by the landmark study on the effector repertoire of Legionella species by Burstein et al., 9 which used genome-wide ortholog clustering to reveal conserved functional modules. Building on this concept, we integrate orthogroup-based presence/absence matrices with machine learning and SHAP (SHapley Additive exPlanations) 10 -based feature importance to prioritize strains likely to possess versatile enzymatic repertoires. Publicly available genomic data (NCBI Entrez) 11 and curated knowledge of known plastic-degrading enzymes ( plasticDB dataset 12 , 13 ) provide a reference framework for interpreting these predictions. The resulting pipeline is modular, reproducible, and interpretable, offering a scalable approach to link conserved gene modules to predicted plastic-degrading potential. This framework can guide experimental validation and support the rational selection of microbial strains for bioremediation applications. 2 Data Selection and Genus Ranking To identify the most promising candidates for plastic degradation, we analyzed the plasticDB dataset, 12 , 13 which catalogs microorganisms and enzymes involved in plastic biodegradation. This resource spans genus, species, and strain levels, providing a comprehensive view of microbial diversity linked to plastic breakdown. We prioritized species represented by many entries in the database, capturing diversity and experimental support, and exhibiting a broad repertoire of plastic-degrading enzymes, indicative of metabolic versatility. To ensure comparability across taxa, these species- and enzyme-level data were aggregated to the genus level, where scores were computed and genera were ranked accordingly. Each genus was assigned a score derived from two weighted components. The first, the species weight , is proportional to the number of species reported for a genus. A larger species count suggests greater diversity and stronger experimental evidence, both of which increase the likelihood of identifying robust plastic degraders. The second, the enzyme weight , is proportional to the number of unique enzymes reported for the genus. A broader enzyme repertoire indicates the potential to degrade different polymer types or to cleave multiple chemical bonds within plastic substrates. As illustrated in Figure 1 , the overall score was calculated as the sum of species and enzyme weights, with contributions normalized across the dataset to ensure comparability. Download figure Open in new tab Figure 1. Visual representation of the scoring system. Species Count and Enzyme Count contribute jointly to the Overall Score. Based on these computed scores, genera were ranked from highest to lowest. Table 1 lists the top 10 genera identified. Among them, Pseudomonas achieved the highest score due to its extensive species diversity and broad enzyme repertoire, followed by genera such as Thermobifida, Cupriavidus , and Bacillus , which showed strong potential but lacked the breadth observed in Pseudomonas . View this table: View inline View popup Download powerpoint Table 1. Top 10 genera from plasticDB ranked by score, based on species and enzyme representation. 2.1 Data Curation and Input Dataset After prioritizing genera, we focused on the top-ranked genus, Pseudomonas , for detailed strain-level analysis. For this purpose, we curated a dataset of 100 complete genomes from NCBI 11 RefSeq and GenBank (accession numbers listed in Supplementary Table S1 ). Only complete or high-quality assemblies were included to ensure reliable comparative genomics. Redundant entries were removed, and representative strains were selected to reflect the natural diversity of Pseudomonas , avoiding overrepresentation of any single species or environment. The curated dataset includes both well-characterized species and unclassified strains labeled as Pseudomonas sp ., providing a broad view of genomic diversity. Genome sizes ranged from 4.4 to 7.2 Mb, consistent with known variability within the genus and indicative of their metabolic adaptability. Notably, 31 strains were unclassified ( Pseudomonas sp .), highlighting the extent of diversity yet to be fully resolved and representing potential sources of novel functional traits. Figure 2 shows the distribution of the 100 curated Pseudomonas strains across 25 species and unclassified lineages. The chart demonstrates a clear imbalance in strain representation: a small number of species, such as Pseudomonas sp . (31 strains), P. syringae (26 strains), P. aeruginosa (9 strains), P. fragariae (7 strains), and P. putida (6 strains), collectively make up nearly 80% of the dataset. These taxa are either clinically significant or environmentally widespread, explaining the higher sequencing coverage. Download figure Open in new tab Figure 2. Distribution of the 100 curated Pseudomonas strains across 25 species. The chart highlights the predominance of a few species and the presence of many rare lineages represented by one or two strains each. In contrast, the remaining 20 species are represented by only one or two strains each. This variability reflects both historical sequencing biases and the natural rarity of certain lineages. Including underrepresented species, despite low strain counts, is critical to capturing the full spectrum of Pseudomonas diversity. The presence of many single-strain species also suggests potential opportunities for discovering novel genes, metabolic pathways, or ecological adaptations that have not yet been extensively studied. Overall, the bar chart highlights two key aspects of the curated dataset: (i) it preserves a robust representation of well-characterized species, allowing meaningful comparative genomics analyses, and (ii) it maintains taxonomic breadth by including rarer lineages, providing avenues for discovering previously uncharacterized genetic or functional diversity. By explicitly visualizing the number of strains per species, Figure 2 offers transparency into the dataset’s composition and informs the interpretation of downstream comparative analyses. 3 Methods Building directly upon the data selection and genus ranking framework described in the previous section, where Pseudomonas was prioritized as the most promising genus due to its high representation of strains and enzyme diversity, we constructed a modular bioinformatics pipeline to predict and evaluate putative plastic-degrading effectors. The pipeline was implemented with a Makefile-based workflow to ensure reproducibility, automated dependency management, and seamless execution across multiple stages. The overall design of the pipeline is illustrated in Figure 3 , which is positioned alongside this section for direct reference. Download figure Open in new tab Figure 3. Pipeline for plastic degrading strain prediction integrating genome analysis with PlasticDB validation. The pipeline begins with scope definition, where either a fixed genus and species or a broader set of taxa are selected based on the ranking system. For the current study, the genus Pseudomonas was chosen due to its dominance in both strain representation and enzyme diversity Table. 1. Genome assemblies were downloaded from NCBI Entrez, 11 ensuring comprehensive coverage across representative species. These assemblies were then subjected to quality assessment using QUAST (Quality Assessment Tool for Genome Assemblies) 17 to evaluate contiguity and assembly statistics, and CheckM 18 to estimate completeness and contamination, thereby ensuring reliable downstream predictions. Following quality control, protein-coding genes were predicted using Prodigal, 14 a tool optimized for bacterial genomes, including incomplete drafts. The predicted proteomes were subsequently clustered into ortholog groups using OrthoFinder 15 , 19 , 20 with Markov Clustering (MCL), which enabled the separation of conserved core proteins shared across all species from accessory proteins that may represent strain-specific innovations. This division was critical for distinguishing effectors with broad conservation from those confined to particular lineages. 3.1 Presence/Absence Matrix Construction Orthogroups, as defined by OrthoFinder, represent sets of genes across bacterial genomes that trace back to a common ancestral gene, thereby capturing conserved evolutionary and functional relationships. A gene belonging to an orthogroup indicates shared ancestry and potential similarity of function with its counterparts across other species. To translate this information into a clinically interpretable form, we constructed a presence/absence matrix ( Figure. 4 ), where each row corresponds to a bacterial strain and each column corresponds to an orthogroup (OG). After preprocessing, this matrix had dimensions of 100 strains × 41 orthogroups , with each cell scored as “1” if the strain contained at least one gene from the orthogroup or “0” if it did not. To bridge genomic content with functional outcomes, we appended an additional column to this dataset containing binary phenotype labels from the PlasticDB resource, which indicate whether each strain has been experimentally associated with plastic degradation activity. Download figure Open in new tab Figure 4. Presence/Absence of Orthogroups Across Genomes The resulting integrated dataset was then used as the input feature space for training a SVM (Support Vector Machine) classifier , enabling systematic modeling of plastic-degrading potential from genomic content. 3.2 Visualization and Integration Framework To interpret the contribution of individual orthogroups to classification outcomes, we employed a Merkmal Treiber Plot (MTP) 16 ( Figure. 6 ). In this visualization, each colored wedge (arc) represents an orthogroup, with its angular width proportional to the number of strains (“carriers”) harboring that feature. Inner radial bands summarize the mean absolute SHAP (SHapley Additive exPlanations) values ( Figure. 5 ) stratified by predicted degrader probability levels (Low, Medium, High), thereby linking orthogroup presence to model confidence. The outer black bars highlight the global importance of each orthogroup across all strains. Legends provide clinical and functional context by identifying the orthogroup, its representative species, the top strain carrying it. This visualization allowed us to prioritize or-thogroups most strongly associated with biodegradation phenotypes, thereby offering both mechanistic interpretability of the classifier and actionable leads for identifying microbial strains with high translational potential in plastic waste remediation. Download figure Open in new tab Figure 5. SHAP 10 top enriched Orthogroups Download figure Open in new tab Figure 6. The MTP plot summarizes the contribution of orthogroups to plastic degradation across Pseudomonas strains. Each wedge corresponds to an orthogroup, with angular width showing how many strains carry it. Nested colored radial bands represent importance across “Low,” “Medium,” and “High” degrader groups: the fill color indicates enzyme functional class, while the outline color indicates whether the feature predicts degrader (red) or non-degrader (blue). The outer black arc shows global importance. This layered view allows rapid identification of which orthogroups, enzymes, and strains are most predictive for plastic degradation. Finally, the distribution of candidate degradative proteins was examined in the context of the core and accessory genome framework. Core determinants represent conserved functions likely central to genus-wide survival, while accessory determinants highlight lineage-specific or niche-adaptive innovations. This combined strategy—integrating orthogroup-level modeling, protein-level annotation, and validation against PlasticDB—provides a comprehensive methodological framework for linking candidate degrader strain prioritization with protein-level discovery of degradation determinants, ultimately supporting the identification of microbial strains with high translational potential for plastic waste remediation. 4 Results Orthogroup Inference and Predictive Overview Orthogroup analysis using OrthoFinder identified a diverse set of gene families across the Pseudomonas genomes included in our study. Among these, a subset of orthogroups was consistently enriched in strains associated with plastic degradation potential. Predictive modeling, evaluated through true and predicted class assignments with associated probabilities, showed high accuracy in distinguishing degrading from non-degrading strains. For example, Pseudomonas strain CP180481 was predicted with a probability of 0.99 to be plastic-degrading, strongly supporting its enzymatic repertoire. Overall, the model demonstrated robust performance, with most positive strains correctly classified and only a limited number of false negatives (e.g., strain CP071658 with a predicted probability of 0.23 despite its known degrading potential). Core and Accessory Orthogroups Relevant to Plastic Degradation From the inferred orthogroups, those associated with enzymes such as hydrolases, esterases, and depolymerases were disproportionately represented in predicted degraders. OG0000784 emerged as the most important predictor with the highest mean absolute SHAP 10 value, suggesting that its encoded functions play a central role in plastic biodegradation. The presence of core orthogroups across multiple Pseudomonas species indicates shared metabolic strategies, while accessory orthogroups contribute to strain-specific versatility. This mirrors evolutionary trends observed in other metabolic systems, where core functions provide baseline activity and accessory functions extend substrate specificity. Functional Annotation of Enzyme Families Annotation of the enriched orthogroups revealed a broad enzymatic spectrum linked to different classes of plastics. Polyhydroxyalkanoate (PHA) and polyhydroxybutyrate (PHB) depolymerases were abundant, reflecting natural adaptation of Pseudomonas to aliphatic polyesters. Cutinases and esterases were linked to polyethylene terephthalate (PET) and polybutylene succinate adipate (PBSA) degradation, while laccases and hydrolases were associated with polyethylene (PE), polystyrene (PS), and polyvinyl chloride (PVC). Polyurethane (PU)-active enzymes, including polyurethane esterases and Impranil-degrading esterases, were detected in specific lineages. This diversity indicates that plastic-degrading Pseudomonas strains rely on a distributed enzymatic toolkit rather than a single universal pathway. Strain-Level Contributions and Key Predictors At the strain level, Pseudomonas sp. CP180479 stood out as the top contributor across multiple orthogroups, reflecting its broad degradation potential. Other strongly predicted degraders included strains CP180480, CP180481, and CP187262 , each supported by high predicted probabilities ( > 0.87). Conversely, strains such as CP071658 and CP071652 were misclassified as non-degraders, likely reflecting either incomplete genome annotations or underrepresentation of accessory orthogroups in the training dataset. Because accessory orthogroups often encode strain-specific functions, missing or partial annotations can obscure their contribution to degradation potential. In contrast, core orthogroups—being widely conserved—are more robust to minor annotation gaps. Nonetheless, predictive accuracy generally requires near-complete genome assemblies, with annotation completeness of ≥95% considered optimal, 21 , 22 and values below 90% carrying substantial risk of misclassification. 21 , 22 These results emphasize the importance of both core orthogroups and strain-specific variations in shaping the predictive landscape, while also emphasizing the need for high-quality, complete genome annotations in such analyses. Model Feature Importance and SHAP Analysis To better understand the contributions of individual features, we employed SHAP (SHapley Additive exPlanations) 10 analysis. The highest-scoring orthogroups, led by OG0000784, 0G0000789, and 0G0003353, were enriched for hydrolases and depolymerases, enzymes directly implicated in polymer breakdown. The distribution of mean absolute SHAP values confirmed that only a small subset of orthogroups explained the majority of predictive performance, highlighting potential marker genes for future biodegradation screening. This hierarchical importance structure aligns with ecological expectations, where a few key enzymes mediate most of the degradation activity, supported by peripheral accessory functions. Integrative Flow of Results Taken together, these findings indicate that Pseudomonas genomes encode a mixture of core and accessory orthogroups that collectively drive plastic degradation. Core sets provide broad-spectrum depolymerase activity, while accessory sets fine-tune substrate range toward PET, PU, PS, PVC, Nylon, and other polymers. The predictive framework, validated by strong agreement between true and predicted degraders, highlights a limited but powerful set of orthogroups as the central drivers of plastic biodegradation potential in Pseudomonas. These results complement the data and methodological framework described earlier, demonstrating that orthogroup-centered approaches can both classify degraders with high precision and uncover the enzymatic basis of plastic degradation diversity. 4.1 Merkmal Treiber Plot (MTP) Analysis: Top Microbeyts (Plastic-Degrading Microbes) To complement the pipeline described earlier, we built the MTP 16 framework in order to visualize and interpret which orthogroups most strongly predict plastic degradation across Pseudomonas strains. Features (orthogroups) were ranked by mean absolute SHAP values, then paired with strain-metadata (species, enzyme annotations, plastics degraded) to generate the MTP, which reveals both the breadth and strength of each feature. In the plot, the circle is divided into wedges (sectors) , each corresponding to an orthogroup. The angular width of a wedge shows how many strains carry that orthogroup. Nested radial colored bands within each wedge display the contribution of that orthogroup across “Low,” “Medium,” and “High” degrader probability groups: the fill color indicates the enzyme functional class (e.g., lipase, hydrolase, esterase), while the band outline color indicates whether the feature pushes the prediction toward degrader (positive, red) or non-degrader (negative, blue). Finally, the outer black arc along the edge of each wedge depicts the global importance of that orthogroup across all strains. Each wedge is labeled “OG | Species | Top-strain,” where OG is the orthogroup, Species is the species carrying that orthogroup, and Top-strain is the carrier strain with the highest model probability. From this MTP analysis, several strongly predictive orthogroups emerged, with actual enzyme–plastic pairings supporting their mechanistic relevance. For example, the orthogroup OG0000784 is associated with Pseudomonas sp ., top strain CP180479, and is linked with enzymes such as lipase, PHA depolymerase, PHB depolymerase and oxidized PVA hydrolase, which in turn correspond to plastics including HDPE, 23 PET, PU, 24 PBSA, 25 PE, 26 – 29 LDPE 30 and PS. 31 The broad substrate range reflected by those plastics mirrors the high importance of that OG in both global and high-probability degrader clusters. Another orthogroup, OG0000855, also associated with Pseudomonas sp ., shows enzymes like polyurethanase, esterases, and hydrolase, and plastics such as PU, LDPE, and HDPE. While its global importance is lower, the MTP plot shows that in “High degrader” probability strains, it contributes conspicuously, suggesting a strong role in strains highly capable of plastic degradation. A further example, OG0002574 in Pseudomonas aeruginosa , carries esterase, lipase, and PHB depolymerase annotations, and correlates with plastics including PET, 32 PU, 33 PE, 29 , 34 and PE blends. Although this OG has moderate global SHAP value, its presence in strain CP187262 and alignment with these enzyme–plastic pairs emphasize its value as a functional marker. The MTP plot thus offers layered insight: it does not merely rank features by statistical importance but integrates biological annotation (which enzymes are involved, which plastics are targeted), directional predictive signal (toward degrader vs. non-degrader), and strain-level metadata (which strain carries that OG most prominently). For clinicians or biotechnologists interested in selecting candidate strains or enzyme systems for experimental validation or bioremediation, MTP provides a guide to which orthogroups may be most promising both globally and in high-activity contexts. 5 Discussion Our study identified top-predicted microbial strains with high plastic-degrading potential through the integration of orthogroup (OG) clustering and SHAP-based feature importance analysis. Orthogroups represent sets of homologous genes shared across strains, and their presence in specific strains indicates conserved genetic modules potentially linked to function. By applying SHAP analysis, we quantified the contribution of each OG to the model’s prediction, thereby highlighting which gene clusters most strongly drive the classification of a strain as a plastic degrader. Strains enriched in high-impact OGs therefore represent prime candidates for further experimental validation, as their genomic content is most predictive of broad enzymatic capabilities against synthetic polymers. Among the top-predicted lineages, Pseudomonas sp . emerged consistently across multiple OGs. In OG0000784, CP180479 was identified as the top strain, representing an OG encompassing 72 strains collected from soil, freshwater, wastewater, and industrial sites. SHAP-ranked OGs in this strain encode enzymes including alkane hydroxylase, esterase, hydrolase, laccase, lipase, PHB depolymerase, PHA depolymerase, polyurethanase, oxidized PVA hydrolase, and PVA dehydrogenase, reflecting a remarkably versatile enzymatic repertoire. These enzymes target a wide variety of plastics, such as HDPE, LDPE, PU, PCL, PET, PHB, PHA, PLA, PS, PVC, O-PVA, and Nylon, highlighting the broad polymer-degrading potential of this lineage. Similarly, OG0000789 and OG0003353 also featured Pseudomonas sp. CP180479 and AP043655 as top representatives, with 63 and 52 strains in their respective OGs. OG0000882, led by Pseudomonas syringae CP180479 , included 68 member strains collected from soil, rhizosphere, and aquatic ecosystems. These top-predicted OGs encode overlapping sets of enzymes, enabling degradation of HDPE, LDPE, PET, PHA, PHB, PU, PVA, PVC, and related polymer blends. The member strains were isolated from diverse environments including soil, marine sediments, land-fills, sewage sludge, and polluted intertidal regions, spanning countries such as Iran, India, USA, Taiwan, Pakistan, China, Japan, South Korea, Nigeria, Thailand, Svalbard, and Serbia. Pseudomonas aeruginosa appeared prominently in OG0002574, OG0000605, OG0002794, and OG0006230, with CP180481 and CP187262 serving as the top representatives. These OGs comprised 9, 23, 10, and 11 member strains respectively, originating from landfills, soils, dumping sites, and culture collections across India, Japan, South Korea, and Iran. The top OGs in these strains encode alkane hydroxylase, esterase, hydrolase, laccase, lipase, PHB depolymerase, PVA dehydrogenase, and various polyurethanases, facilitating degradation of plastics including HDPE, LDPE, PU, PBSA, PHB, PET, PS, and PU blends. The observed distribution of top-predicted species and strains illustrates the polygenic nature of plastic degradation, with multiple enzyme classes acting synergistically across diverse polymer backbones. Membership in each OG implies that all strains share conserved genetic modules, suggesting potential for plastic-degrading function even in less-characterized strains. The combination of orthogroup conservation and SHAP-driven feature importance provides an interpretable framework linking genomic content to predicted plastic-degrading function. The ecological and geographic diversity of these strains, coupled with the number of strains per OG, emphasizes the importance of environmental context in shaping degradative capabilities and supports their selection for laboratory validation, bioremediation trials, and the rational design of microbial consortia for targeted polymer breakdown in polluted environments. Details of all strains within each top-predicted OG, are provided in Supplementary Table S2 . 6 Limitations This study has limitations that should be considered when interpreting the results. First, presence/absence matrices rely on accurate orthogroup assignments, which can be influenced by genome completeness and annotation quality within the genus. Second, limited representation of certain clades may reduce clustering resolution and affect the robustness of inferred relationships. Third, the analysis captures genomic potential rather than experimentally confirmed functions, so observed patterns may not directly translate to phenotypic traits. Finally, as this study is focused on a single genus, the findings and conclusions may not be generalizable to other bacterial taxa. 7 Conclusion and Future Work This study presents a modular and interpretable pipeline for systematically linking genomic content to predicted plastic-degrading potential in microbial strains. By integrating orthogroup clustering with SHAP-based feature importance analysis, the framework identifies conserved gene modules that contribute most strongly to predictive models, highlighting strains with the highest likelihood of enzymatic activity against diverse plastics. The combination of presence/absence matrices, clustering, and interpretable model outputs provides a transparent and reproducible approach for prioritizing candidate strains for experimental validation. Application of this pipeline to a single genus revealed biologically meaningful patterns, with top-predicted strains such as Pseudomonas sp . consistently harboring orthogroups encoding enzymes capable of degrading a broad spectrum of polymers, including HDPE, LDPE, PET, PU, PHA, PHB, PLA, PS, PVC, and PVA. The analysis demonstrated the polygenic nature of plastic degradation, where multiple enzyme classes act synergistically, and conserved orthogroups allow inference of degradative potential even in less-characterized strains. The geographic and ecological diversity of these strains further emphasizes the importance of environmental context in shaping enzymatic capabilities. Future Work Several directions can extend the biological and computational scope of this work. Experimental validation of top-ranked strains and orthogroups remains a priority to confirm predicted enzymatic activity. Expanding the pipeline to additional strains within the genus and closely related taxa will improve the resolution of clustering and enhance generalizability. Future annotation enrichment should leverage tools such as InterProScan 35 – 37 and Pfam 38 to systematically identify conserved protein domains, motifs, and catalytic sites, enabling the derivation of multidimensional feature sets—including amino acid composition, GC content, and domain architectures—for more precise functional inference. Incorporating structural modeling and pathway-level annotation (e.g., KEGG, PlasticDB) could further refine predictions and provide mechanistic insight into substrate specificity. Ensemble learning approaches may improve predictive accuracy, reducing uncertainty in orthogroup-based inference. Finally, development of interactive visualizations will facilitate exploration of strain-specific predictions, orthogroup conservation patterns, and SHAP-derived feature importance, supporting translational applications in environmental biotechnology and microbial ecology. Final Remarks By systematically combining orthogroup clustering, interpretable predictive modeling, and curated annotation, the pipeline offers a robust framework for discovering microbial strains and gene modules with potential plastic-degrading activity. This approach bridges computational prediction with biological insight, enabling high-throughput prioritization of candidate strains and providing actionable guidance for experimental validation, bioremediation strategies, and the rational design of microbial consortia for polymer degradation. Funding The authors received no specific funding for this work. Conflicts of Interest The authors declare no competing interests. Author Contributions -Dr. Lokendra S. Thakur conceived the idea, designed the study, developed method and algorithm, curated data and done analysis, developed pipeline, all sections writing. -Dr. Gurpreet Bharj performed the microbiological and clinical analysis. -Manish Saroya checked citations and contributed in refining content. -All authors reviewed and approved the final version. Ethics Approval and Consent to Participate Not applicable. Data Availability We accessed raw data from the publicly available repository: https://www.ncbi.nlm.nih.gov/ Code supporting this study are available on reasonable request. Disclaimer Preprints are preliminary reports that have not been peer reviewed. They should not be regarded as conclusive, guide clinical practice, or be reported in news media as established information. Supplementary View this table: View inline View popup Supplementary Table S1. Curated Pseudomonas genome dataset from NCBI RefSeq/GenBank. View this table: View inline View popup Supplementary Table S2. Features, strains, enzymes, plastics, and SHAP values identified in Pseudomonas species . Acknowledgements Data used in the preparation of this article were obtained from publicly available nucleotide and protein sequences in the National Center for Biotechnology Information (NCBI) RefSeq and GenBank 39 databases. RefSeq provides curated, non-redundant reference sequences, 40 and GenBank contains publicly submitted sequences. 41 Accession numbers for all sequences used are provided in Supplementary Table S1 . We are grateful to Dr. Eyal Y. Kimchi (Department of Neurology, Northwestern University Feinberg School of Medicine, USA), Dr. Strajit Ghosh (Massachusetts General Hospital and Harvard Medical School, USA), Dr. Fernando Gómez-Baquero (Jacobs Technion–Cornell Institute, Cornell Tech; NSF Upstate New York Energy Storage Engine, USA), Dr. Kamana Porwal (Department of Mathematics, IIT Delhi, India), and Dr. Mustafa Hajij (MSDSAI Program, University of San Francisco, California, USA) for their valuable guidance, support, and insightful suggestions, which greatly contributed to this work. Footnotes https://www.ncbi.nlm.nih.gov/ http://plasticdb.org . References [1]. ↵ United Nations Environment Programme . Plastic pollution facts . https://www.unep.org/plastic-pollution , 2023 . Accessed: 2025-09-13 . [2]. ↵ RTS . Plastic pollution in the ocean: Facts and statistics . https://www.rts.com/blog/plastic-pollution-in-the-ocean-facts-and-statistics/ , 2023 . Accessed: 2025-09-13 . [3]. ↵ The Guardian . Just 9.5% of plastic made in 2022 used recycled material, study shows . https://www.theguardian.com/environment/2025/apr/10/just-95-of-plastic-made-in-2022-used-recycled-material-study-shows , 2025 . Accessed: 2025-09-13 . [4]. ↵ End Plastic Waste . Plastic waste management framework . https://www.endplasticwaste.org/insights/reports/plastic-waste-management-framework , 2024 . Accessed: 2025-09-13 . [5]. ↵ EKB Roman , MA Ramos , G Tomazetto , BB Foltran , M. Galvão , I Ciancaglini , R Tramontina , F de Almeida Rodrigues , LS da Silva , ALH Sandano , DGDS Fernandes , DV Almeida , D. Baldo , JM de Oliveira Junior , W Garcia , A Damasio , and FM Squina . Plastic-degrading microbial communities reveal novel microorganisms, pathways, and biocatalysts for polymer degradation and bioplastic production . Science of The Total Environment , 949 : 174876 , 2024 . Erratum in: Sci Total Environ. 2024 Nov 20;952:175678. doi: 10.1016/j.scitotenv.2024.175678 . OpenUrl CrossRef PubMed [6]. ↵ Sophie A. Howard and Ronan R. McCarthy . Modulating biofilm can potentiate activity of novel plastic-degrading enzymes . npj Biofilms and Microbiomes , 9 ( 1 ): 72 , 2023 . OpenUrl [7]. ↵ K. Houssini , J. Li , and Q. Tan . Complexities of the global plastics supply chain revealed in a trade-linked material flow analysis . Communications Earth & Environment , 6 : 257 , 2025 . OpenUrl [8]. ↵ Ping Wang and Yu-zhen Shi . From waste to opportunity: The potential of microbial biodegradation in plastic pollution mitigation . Journal of Hazardous Materials Advances , 19 : 100865 , 2025 . OpenUrl [9]. ↵ David Burstein , Fabiana Amaro , Tal Zusman , Zohar Lifshitz , Oren Cohen , Julie A Gilbert , Tal Pupko , Howard A Shuman , Gil Segal , and Michal Rasis . Genomic analysis of 38 legionella species identifies large and diverse effector repertoires . Nature Genetics , 48 ( 2 ): 167 – 175 , 2016 . OpenUrl CrossRef PubMed [10]. ↵ Scott M Lundberg and Su-In Lee . A unified approach to interpreting model predictions . In Advances in neuralinformation processing systems , volume 30 , 2017 . [11]. ↵ National Center for Biotechnology Information (NCBI) . Ncbi genome database . https://www.ncbi.nlm.nih.gov/genome/ , 2025 . Accessed: 2025-09-13 . [12]. ↵ Victor Gambarini , Olga Pantos , Joanne M Kingsbury , Louise Weaver , Kim M Handley , and Gavin Lear . Plasticdb: a database of microorganisms and proteins linked to plastic biodegradation . Database , 2022 : baac008 , 2022 . OpenUrl CrossRef PubMed [13]. ↵ Jan Zrimec , Monika Kokina , Karl H Jonsson , Francisco Zorrilla , and Aleksej Zelezniak . Plasticdb: a database of microorganisms and enzymes involved in plastic biodegradation . Nucleic Acids Research , 48 ( D1 ): D1076 – D1084 , 2020 . OpenUrl PubMed [14]. ↵ Doug Hyatt , Gwo-Liang Chen , Phillip F LoCascio , Miriam L Land , Frank W Larimer , and Loren J Hauser . Prodigal: prokaryotic gene recognition and translation initiation site identification . BMC Bioinformatics , 11 ( 1 ): 119 , 2010 . OpenUrl CrossRef PubMed [15]. ↵ David M Emms and Steven Kelly . Orthofinder: solving fundamental biases in whole genome comparisons dramatically improves orthogroup inference accuracy . Genome Biology , 16 ( 1 ): 157 , 2015 . OpenUrl CrossRef PubMed [16]. ↵ Lokendra S. Thakur , Gurpreet Bharj , Lokesh Sangabattula , and Bushra Malik . ci-fgbd: Cluster-integrated fast generalized bruhat decomposition for multimodal data clustering in alzheimer’s disease . medRxiv , 2025 . [17]. ↵ Alexey Gurevich , Vladislav Saveliev , Nikolay Vyahhi , and Glenn Tesler . Quast: quality assessment tool for genome assemblies . Bioinformatics , 29 ( 8 ): 1072 – 1075 , 2013 . OpenUrl CrossRef PubMed Web of Science [18]. ↵ Donovan H Parks , Michael Imelfort , Connor T Skennerton , Philip Hugenholtz , and Gene W Tyson . Checkm: assessing the quality of microbial genomes recovered from isolates, single cells, and metagenomes . Genome Research , 25 ( 7 ): 1043 – 1055 , 2015 . OpenUrl Abstract / FREE Full Text [19]. ↵ David M Emms and Steven Kelly . Orthofinder: phylogenetic orthology inference for comparative genomics . Genome Biology , 20 ( 1 ): 238 , 2019 . OpenUrl CrossRef PubMed [20]. ↵ David M Emms and Steven Kelly . Stride: species tree root inference from gene duplication events . Molecular Biology and Evolution , 34 ( 12 ): 3267 – 3278 , 2017 . OpenUrl CrossRef PubMed [21]. ↵ Donovan H. Parks , Michael Imelfort , Connor T. Skennerton , Philip Hugenholtz , and Gene W. Tyson . Checkm: assessing the quality of microbial genomes recovered from isolates, single cells, and metagenomes . Genome Research , 25 ( 7 ): 1043 – 1055 , 2015 . OpenUrl Abstract / FREE Full Text [22]. ↵ Felipe A. Simão , Robert M. Waterhouse , Panagiotis Ioannidis , Evgenia V. Kriventseva , and Evgeny M. Zdobnov . Busco: assessing genome assembly and annotation completeness with single-copy orthologs . Bioinformatics , 31 ( 19 ): 3210 – 3212 , 2015 . OpenUrl CrossRef PubMed [23]. ↵ V. Balasubramanian , K. Natarajan , B. Hemambika , N. Ramesh , C. S. Sumathi , R. Kottaimuthu , and V. Rajesh Kannan . High-density polyethylene (hdpe)-degrading potential bacteria from marine ecosystem of gulf of mannar, india . Letters in Applied Microbiology , 51 ( 2 ): 205 – 211 , 2010 . OpenUrl PubMed [24]. ↵ B. W. Stamps , S. Zingarelli , C. S. Hung , C. A. Drake , V. A. Varaljay , B. S. Stevenson , and W. J. Crookes-Goodson . Finished genome sequence of a polyurethane-degrading pseudomonas isolate . Genome Announcements , 6 ( 9 ): e00084 – 18 , 2018 . OpenUrl [25]. ↵ A. K. Urbanek , W. Rymowicz , M. C. Strzelecki , W. Kociuba , Ł. Franczak , and A. M. Mirończuk . Isolation and characterization of arctic microorganisms decomposing bioplastics . AMB Express , 7 ( 1 ): 148 , 2017 . OpenUrl CrossRef PubMed [26]. ↵ K. Kathiresan . Polythene and plastics-degrading microbes from the mangrove soil . Revista de Biologia Tropical , 51 ( 3-4 ): 629 – 633 , 2003 . OpenUrl PubMed [27]. S. Satyalakshmi . Isolation and identification of polythene bags degrading bacteria from visakhapatnam dumping yard . International Journal of Pharmaceutical Sciences and Research , 7 ( 10 ): 4200 – 4205 , 2016 . OpenUrl [28]. S. Nanda and S. S. Sahu . Biodegradability of polyethylene by brevibacillus, pseudomonas, and rhodococcus spp . New York Science Journal , 3 ( 7 ): 95 – 98 , 2010 . OpenUrl [29]. ↵ H. Shahreza , A. A. Sepahy , F. Hosseini , and R. K. Nejad . Molecular identification of pseudomonas strains with polyethylene degradation ability from soil and cloning of alkb gene . Archives of Pharmacy Practice , 10 ( 4 ): 1 – 7 , 2019 . OpenUrl [30]. ↵ R. Usha , T. Sangeetha , and M. Palaniswamy . Screening of polyethylene degrading microorganisms from garbage soil . Libyan Agriculture Research Center Journal International , 2 ( 4 ): 200 – 204 , 2011 . OpenUrl [31]. ↵ A. J. Mohan , V. C. Sekhar , T. Bhaskar , and K. M. Nampoothiri . Microbial assisted high impact polystyrene (hips) degradation . Bioresource Technology , 213 : 204 – 207 , 2016 . OpenUrl CrossRef PubMed [32]. ↵ F. Babazadeh , S. Gharavi , M. R. Soudi , M. Zarrabi , and Z. Talebpour . Potential for polyethylene terephthalate (pet) degradation revealed by metabarcoding and bacterial isolates from soil around a bitumen source in southwestern iran . Journal of Polymers and the Environment , 31 ( 4 ): 1279 – 1291 , 2023 . OpenUrl [33]. ↵ Z. Shah , F. Hasan , L. Krumholz , D. F. Aktas , and A. A. Shah . Degradation of polyester polyurethane by newly isolated pseudomonas aeruginosa strain mza-85 and analysis of degradation products by gc–ms . International Biodeterioration & Biodegradation , 77 : 114 – 122 , 2013 . OpenUrl CrossRef [34]. ↵ S. Ali , A. Rehman , S. Z. Hussain , and D. A. Bukhari . Characterization of plastic degrading bacteria isolated from sewage wastewater . Saudi Journal of Biological Sciences , 30 ( 5 ): 103628 , 2023 . OpenUrl PubMed [35]. ↵ E. M. Zdobnov and R. Apweiler . Interproscan — an integration platform for the signature-recognition methods in interpro . Bioinformatics , 17 ( 9 ): 847 – 848 , 2001 . OpenUrl CrossRef PubMed Web of Science [36]. E. Quevillon , V. Silventoinen , S. Pillai , N. Harte , N. Mulder , R. Apweiler , and R. Lopez . Interproscan: protein domains identifier . Nucleic Acids Research , 33 ( Web Server issue ): W116 – W120 , 2005 . OpenUrl CrossRef PubMed Web of Science [37]. ↵ P. Jones , D. Binns , H.-Y. Chang , M. Fraser , W. Li , C. McAnulla , H. McWilliam , J. Maslen , A. Mitchell , G. Nuka , et al. Interproscan 5: genome-scale protein function classification . Bioinformatics , 30 ( 9 ): 1236 – 1240 , 2014 . OpenUrl CrossRef PubMed Web of Science [38]. ↵ Jaina Mistry , Sara Chuguransky , Lowri Williams , Matloob Qureshi , Gustavo A. Salazar , Erik L. L. Sonnhammer , Silvio C. E. Tosatto , Lisanna Paladin , Shriya Raj , Lorna J. Richardson , Robert D. Finn , and Alex Bateman . Pfam: The protein families database in 2021 . Nucleic Acids Research , 49 ( D1 ): D412 – D419 , 2021 . OpenUrl CrossRef PubMed [39]. ↵ Eric W. Sayers , Mark Cavanaugh , Karen Clark , James Ostell , Kim D. Pruitt , and Ilene Karsch-Mizrachi . Genbank . Nucleic Acids Research , 47 ( D1 ): D94 – D99 , 2019 . OpenUrl CrossRef PubMed [40]. ↵ N. O’Leary , MW Wright , JR Brister , et al. Reference sequence (refseq) database at ncbi: current status, taxonomic expansion, and functional annotation . Nucleic Acids Research , 44 ( D1 ): D733 – D745 , 2016 . OpenUrl CrossRef PubMed [41]. ↵ DA Benson , M Cavanaugh , K Clark , et al. Genbank . Nucleic Acids Research , 41 ( Database issue ): D36 – D42 , 2013 . OpenUrl CrossRef PubMed Web of Science View the discussion thread. Back to top Previous Next Posted September 19, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Interpretable Machine Learning and Comparative Genomics Reveal Microbial Plastic-Degrading (Microbeyt) Potential Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Interpretable Machine Learning and Comparative Genomics Reveal Microbial Plastic-Degrading (Microbeyt) Potential Lokendra S. Thakur , Gurpreet Bharj , Manish Saroya bioRxiv 2025.09.17.676701; doi: https://doi.org/10.1101/2025.09.17.676701 Share This Article: Copy Citation Tools Interpretable Machine Learning and Comparative Genomics Reveal Microbial Plastic-Degrading (Microbeyt) Potential Lokendra S. Thakur , Gurpreet Bharj , Manish Saroya bioRxiv 2025.09.17.676701; doi: https://doi.org/10.1101/2025.09.17.676701 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genomics Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17690) Bioengineering (13892) Bioinformatics (41936) Biophysics (21451) Cancer Biology (18588) Cell Biology (25499) Clinical Trials (138) Developmental Biology (13378) Ecology (19899) Epidemiology (2067) Evolutionary Biology (24320) Genetics (15609) Genomics (22506) Immunology (17736) Microbiology (40394) Molecular Biology (17181) Neuroscience (88603) Paleontology (666) Pathology (2832) Pharmacology and Toxicology (4824) Physiology (7641) Plant Biology (15152) Scientific Communication and Education (2045) Synthetic Biology (4294) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00