Predicting the pathway involvement of metabolites annotated in the MetaCyc knowledgebase

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

The associations of metabolites with biochemical pathways are highly useful information for interpreting molecular datasets generated in biological and biomedical research. However, such pathway annotations are sparse in most molecular datasets, limiting their utility for pathway level interpretation. To address these shortcomings, several past publications have presented machine learning models for predicting the pathway association of small biomolecule (metabolite and zenobiotic) using data from the Kyoto Encyclopedia of Genes and Genomes (KEGG). But other similar knowledgebases exist, for example MetaCyc, which has more compound entries and pathway definitions than KEGG. As a logical next step, we trained and evaluated multilayer perceptron models on compound entries and pathway annotations obtained from MetaCyc. From the models trained on this dataset, we observed a mean Matthews correlation coefficient (MCC) of 0.845 with 0.0101 standard deviation, compared to a mean MCC of 0.847 with 0.0098 standard deviation for the KEGG dataset. These performance results are pragmatically the same, demonstrating that MetaCyc pathways can be effectively predicted at the current state-of-the-art performance level. Author summary Many thousands of different molecules play important roles in the processes of life. To generally handle the complexity of life, biological and biomedical researchers typically organize the molecular parts and pieces of biological processes into pathways of biomolecules and their myriad of molecular interactions. While the role of large macromolecules like proteins are well characterized within these pathways, the role of small biomolecules are not as comprehensively known. To close this knowledge gap, several machine learning models have been trained on data from a knowledgebase known as the Kyoto Encyclopedia of Genes and Genomes (KEGG) to predict which pathways a small biomolecule is associated with. More data generally improves these machine learning models. So in this work, we used the MetaCyc knowledgebase to increase the amount of data available by about ten-fold and then trained new machine learning models that demonstrate comparable prediction performance to models trained on KEGG, but covering 8-fold more pathways defined in MetaCyc vs KEGG.
Full text 34,274 characters · extracted from preprint-html · click to expand
Predicting the pathway involvement of metabolites annotated in the MetaCyc knowledgebase | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Predicting the pathway involvement of metabolites annotated in the MetaCyc knowledgebase View ORCID Profile Erik D. Huckvale , View ORCID Profile Hunter N.B. Moseley doi: https://doi.org/10.1101/2024.10.29.620954 Erik D. Huckvale 1 Markey Cancer Center, University of Kentucky , Lexington, KY, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Erik D. Huckvale Hunter N.B. Moseley 1 Markey Cancer Center, University of Kentucky , Lexington, KY, USA 2 Superfund Research Center, University of Kentucky , Lexington, KY, USA 3 Department of Toxicology and Cancer Biology, University of Kentucky , Lexington, KY, USA 4 Department of Molecular and Cellular Biochemistry, University of Kentucky , Lexington, KY, USA 5 Institute for Biomedical Informatics, University of Kentucky , Lexington, KY, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Hunter N.B. Moseley For correspondence: kenneth.kim{at}unil.ch Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract The associations of metabolites with biochemical pathways are highly useful information for interpreting molecular datasets generated in biological and biomedical research. However, such pathway annotations are sparse in most molecular datasets, limiting their utility for pathway level interpretation. To address these shortcomings, several past publications have presented machine learning models for predicting the pathway association of small biomolecule (metabolite and zenobiotic) using data from the Kyoto Encyclopedia of Genes and Genomes (KEGG). But other similar knowledgebases exist, for example MetaCyc, which has more compound entries and pathway definitions than KEGG. As a logical next step, we trained and evaluated multilayer perceptron models on compound entries and pathway annotations obtained from MetaCyc. From the models trained on this dataset, we observed a mean Matthews correlation coefficient (MCC) of 0.845 with 0.0101 standard deviation, compared to a mean MCC of 0.847 with 0.0098 standard deviation for the KEGG dataset. These performance results are pragmatically the same, demonstrating that MetaCyc pathways can be effectively predicted at the current state-of-the-art performance level. Author summary Many thousands of different molecules play important roles in the processes of life. To generally handle the complexity of life, biological and biomedical researchers typically organize the molecular parts and pieces of biological processes into pathways of biomolecules and their myriad of molecular interactions. While the role of large macromolecules like proteins are well characterized within these pathways, the role of small biomolecules are not as comprehensively known. To close this knowledge gap, several machine learning models have been trained on data from a knowledgebase known as the Kyoto Encyclopedia of Genes and Genomes (KEGG) to predict which pathways a small biomolecule is associated with. More data generally improves these machine learning models. So in this work, we used the MetaCyc knowledgebase to increase the amount of data available by about ten-fold and then trained new machine learning models that demonstrate comparable prediction performance to models trained on KEGG, but covering 8-fold more pathways defined in MetaCyc vs KEGG. Introduction Metabolism is the set of biochemical reactions and related processes for sustaining life. These biochemical reactions convert environmentally-derived and endogenous molecular and energy resources into useful forms of chemical energy and molecular building blocks that drive biological processes, while converting metabolic wastes into forms that can be disposed of. Metabolism can also be viewed as the entirety of these biochemical reactions organized as a metabolic network. Related biological processes can be included into a more general molecular interaction network that is centered on metabolism. Such networks are representable as a hypergraph with biomolecules as nodes and chemical reactions and non-covalent molecular interactions as edges. Certain connected parts or subgraphs of the metabolic-centric molecular interaction network are then defined as distinct “pathways” ( 1 – 3 ). Compounds connected within a pathway are defined as being associated with that pathway, which is typically represented as a pathway annotation for the compound. Pathway annotations are highly useful for mechanistic interpretation of complex molecular datasets. These annotations are often used in pathway annotation enrichment analysis to identify perturbed pathways based on large datasets of perturbed biomolecules, providing pathway-level insight and interpretation of such datasets. While knowledgebases such as the Kyoto Encyclopedia of Genes and Genomes (KEGG) ( 4 ) and MetaCyc ( 5 ) contain useful pathway annotations for several thousand compounds, the majority of detected compounds in metabolomics datasets have no pathway annotations, limiting their use in pathway annotation enrichment analysis. Considering the costly and cumbersome work involved in experimentally determining these annotations in vitro, many researchers could benefit from predicting pathway annotations for compounds in silico. Several prior publications present machine learning models that perform this pathway prediction task, trained and tested with data from KEGG ( 6 – 9 )( 10 ). Based on the success of these models trained on KEGG-derived datasets, a logical next step is to attempt constructing a similar machine learning dataset from MetaCyc compounds and pathway annotations. The MetaCyc dataset focuses on metabolic pathways as compared to KEGG which contains a wider variety of types of pathways: only 184 out of 502 pathways in KEGG are metabolic pathways. Moreover, MetaCyc contains far more pathway and compound entries. In this work, we demonstrate that MetaCyc pathways can be predicted as effectively as KEGG pathways. Materials and methods The construction of a dataset for model training and evaluation requires the molecular structures of the compounds along with their pathway annotations. Our molecular structure parsing methods expect molfiles ( 11 ). On July 9 th 2024, we downloaded the molfiles from MetaCyc’s website here https://metacyc.org/download.shtml . The MetaCyc pathways, similar to KEGG, are organized into a hierarchy as seen here: https://metacyc.org/META/class-tree?object=Pathways . The compound to pathway mappings were downloaded on July 9 th 2024. We accessed them by traversing the pathway hierarchy using MetaCyc’s web API at the following base URL: https://biocyc.org/META/ajax-direct-subs. With the pathway ID query parameter, you can get the children of a given pathway node. Starting with the base ID ‘Pathways’, the first requestion URL is https://biocyc.org/META/ajax-direct-subs?object=Pathways . Next, you obtain children pathway IDs and recursively request children pathways. Traversing the hierarchy in this way enables access to all the MetaCyc pathway IDs from the root node down to the leaf nodes. Compound associations were obtained from MetaCyc’s web API using this base URL: https://websvc.biocyc.org/apixml?fn=compounds-of-pathway . Again, a pathway ID query parameter is specified. For example, the URL https://websvc.biocyc.org/apixml?fn=compounds-of-pathway&id=META:PWY-7723&detail=low obtains the compounds associated with the pathway corresponding to pathway ID META:PWY-7723. After obtaining the MetaCyc molfiles and compound-pathway mappings, we constructed the MetaCyc dataset with the methods used for the KEGG dataset ( 10 ). This involved creating feature vectors representing compounds, feature vectors representing pathways, and concatenating them together in a cross join. The resulting compound- pathway paired feature vector is then concatenated with a boolean label indicating whether the given compound is associated with the given pathway ( 8 ). We created the compound features using an atom coloring technique introduced in our lab ( 7 , 12 – 14 ). The atom colors represent molecular substructures 1, 2, or 3 bonds away from a central atom node and the feature values are the number of times that substructure appears in the compound. With these atom color counts acting as features for an individual compound, the pathway features are an aggregation of the features of the compounds associated with a given pathway ( 8 ). Both the compound features and pathway features were deduplicated and then normalized ( 10 ). Table 1 compares the MetaCyc dataset to the KEGG dataset. We see that with nearly 40 million entries, the MetaCyc-derived dataset is the largest published to date, having more than 10-fold the number of paired compound-pathway entries than the KEGG- derived dataset. This is mostly due to the 8-fold larger number of pathway definitions in MetaCyc vs KEGG. View this table: View inline View popup Download powerpoint Table 1 Comparing the KEGG dataset to the MetaCyc dataset. With the pathways being hierarchically organized, there are pathways at different hierarchical levels. We will refer to the top-level pathways, e.g. ‘Biosynthesis’, ‘Detoxification’, ‘Glycan Pathways’, ‘Transport’, etc. as level 1 or L1. Likewise, we refer to the level 2 pathways as L2, and the remaining as L3, L4, L5, L6, L7, and L8. To determine the impact of excluding certain pathway levels from the training set on the performance of the remaining pathways, we constructed two subsets of the full MetaCyc dataset i.e. L2+ and L3+. The L1+ dataset contains all of the pathways in the hierarchy, while the L2+ excludes the L1 pathways and contains the L2 pathways and all pathways underneath them in the hierarchy. Likewise, the L3+ dataset excludes the L1 and L2 pathways. Table 2 shows stats for each of these datasets. View this table: View inline View popup Download powerpoint Table 2 Comparing the subsets of the MetaCyc dataset where L1+ is the full MetaCyc dataset, L2+ excludes L1 pathways, and L3+ excludes L1 and L2 pathways. For our analyses related to the hierarchy levels, we rolled L8 into L7 to form the L7+L8 hierarchy level for more statistically meaningful results, since L8 only had 11 pathways. The reason we did not roll the L1 pathways into L2, is because the L1 pathways are much larger than the L8 pathways, even though there are only 12 L1 pathways. Fig 1 shows the number of pathways in each hierarchy level from L1 to L7+L8. Fig S1 shows this same info, but it shows the number of L7 and L8 pathways separately. Download figure Open in new tab Fig 1 The number of pathways within each hierarchy level. We used a cross validation (CV) analysis with characteristics of both a bootstrap and jackknife analysis, involving 200 or 50 separate iterations of 10-fold cross validation. On each CV iteration, we split the dataset into a train and test set in a stratified manner across 10 folds ( 15 ), ensuring that the proportion of positive entries in the train set is as close as possible to the proportion in the test set. We trained a multi-layer perceptron binary classifier on the train set composed of 9 folds and evaluated on the test set composed of a single fold, collecting the number of true positives (TP), true negatives (TN), false positives (FP), and false negatives (FN). This enabled us to calculate metrics including the Matthews correlation coefficient (MCC) ( 16 ) on each CV iteration and then to calculate a mean, median, and standard deviation of each metric across CV iterations. It also enabled us to calculate the MCC per compound and per pathway by counting the TP, TN, FP, and FN in the test set for a given compound or pathway (since each entry is a compound-pathway pair) and then summing those counts across all CV iterations. An MCC cannot be calculated on a single CV iteration since there may not be enough positive entries in the test set for a single compound or pathway to avoid a division by zero. However, we can calculate an overall MCC for an individual compound or pathway by summing TP, TN, FP, and FN across all CV iterations. Likewise, we calculated the MCC for each hierarchy level by summing the TP, TN, FP, and FN across all pathways in a given hierarchy level across all CV iterations. We calculated the overall MCCs of individual compounds and pathways from the L1+ dataset (i.e. the full dataset) and ran the L1+ dataset for 200 CV iterations. We ran the L2+ and L3+ datasets for 50 CV iterations and used the results from those datasets to compare overall MCC across different hierarchy levels. We ran the L1+ dataset for more iterations to ensure a valid MCC calculation for individual compounds or pathways, some of which might not have many positive entries, necessitating more CV iterations to have enough TP or FP. The hardware used for this work included machines with up to 2 terabytes (TB) of random-access memory (RAM) and central processing units (CPUs) of 3.8 gigahertz (GHz) of processing speed. The name of the CPU chip was ‘Intel(R) Xeon(R) Platinum 8480CL’. The graphic processing units (GPUs) used had 81.56 gigabytes (GB) of GPU RAM, with the name of the GPU card being ‘NVIDIA H100 80GB HBM3’. All code for this work was written in major version 3 of the Python programming language ( 17 ). Data processing and storage was done using the Pandas ( 18 ), NumPy ( 19 ), and H5Py ( 20 ) packages. Models were constructed and trained using the PyTorch Lightning ( 21 ) package built upon the PyTorch ( 22 ) package. The stratified train test splits were computed using the Sci-Kit Learn package ( 23 ). Results were initially stored in an SQL database ( 24 ) using the DuckDB ( 25 ) package. Results were processed and visualized using Jupyter notebooks ( 26 ), the seaborn package ( 27 ) built upon the MatPlotLib ( 28 ) package, and the Tableau business intelligence application ( 29 ). Results Main results Table 3 includes the mean, median, and standard deviation of the MCC calculated across CV iterations of the L1+ (full), L2+, and L3+ datasets. The mean and median MCCs are very close, even though distributions of MCC do not look unimodal nor very symmetric. Table S1 shows these results for other metrics i.e. accuracy, precision, recall, F1 score, and specificity. We observe a decrease in performance when we exclude higher level pathways. View this table: View inline View popup Download powerpoint Table 3 Matthew’s correlation coefficient statistics for models trained on the L1+, L2+, and L3+ datasets. Fig 2 shows the distribution of MCC across the 200 CV iterations of the L1+ dataset. Download figure Open in new tab Fig 2 Distribution of the MCC across all CV iterations on the L1+ (full) MetaCyc dataset. Fig 3 shows the overall MCC across pathways at certain hierarchy levels and across all CV iterations of each dataset i.e. L1+, L2+, and L3+. We see that performance consistently declines at deeper hierarchy levels. Download figure Open in new tab Fig 3 Comparing the overall MCC of pathway hierarchy levels within each dataset. Fig 4 shows the same information as Fig 3 , but compares performance across the datasets within hierarchy levels rather than the performance of hierarchy levels within the datasets. We see that inclusion of the L1 pathways in the L1+ dataset improved the performance of the L2 pathways. Similarly, inclusion of the L2 pathways in the L2+ dataset improved performance of the L3 pathways, though inclusion of the L1 pathways did not provide further improvement of the L3 pathways. Meanwhile the pathways in the L4, L5, L6, and L7+L8 hierarchy levels did not experience significant change in performance regardless of whether the L1 or L2 pathways were included in training. Download figure Open in new tab Fig 4 Comparing the overall MCC between datasets within each hierarchy level. MCC and compound/pathway size We define compound size as the number of non-hydrogen atoms in the compound. We define pathway size as the sum of the sizes of the compounds associated with that pathway. Fig 5 shows the distribution of the size of compounds and pathways in the full MetaCyc dataset. Download figure Open in new tab Fig 5 Distribution of compound size (number of non-hydrogen atoms in the molecule) and pathway size (sum of the sizes of compounds associated with a pathway). Distribution of Compound and Pathway Size While Fig 5 shows the distribution of the size of all pathways, Fig 6 shows the distribution of pathway size within each hierarchy level. We see an overall trend of pathway size declining deeper into the hierarchy. Note that the y-axis is on a log scale to illustrate the wide range of pathway sizes that span multiple orders of magnitude. Download figure Open in new tab Fig 6 Violinplot showing the distribution of the sizes of the pathways within each hierarchy level. Fig 7 shows the distribution of the overall MCC of individual pathways and compounds. There were 9,847 compounds and 4,055 pathway entries in the full MetaCyc dataset. The MCC distributions are negatively skewed for both pathways and compounds, with the compounds highly negatively skewed, likely due to truncation effects from boundary conditions. Download figure Open in new tab Fig 7 Distribution of the MCC of individual pathways and compounds. Distribution of Pathway and Compound MCC Fig 8 contains scatterplots comparing the MCC of pathways and compounds to their size. Fig 8a and Fig 8c compare pathway size to MCC and compound size to MCC respectively, while Fig 8b and Fig 8d are the same plots but with the x axis on a log scale for better visibility. We see that the pathway to MCC plot exhibits a funnel shape indicating that the variance of pathway MCC decreases as pathway size increases. Additionally, we see that the maximum MCC is lower for smaller pathways. Likewise, the maximum compound MCC does not reach 1.0 until reaching a compound size of 7 and we see before that, maximum compound MCC decreases as compound size decreases. Download figure Open in new tab Fig 8 Scatterplots comparing the size of pathways and compounds to their MCC. Comparing MetaCyc to KEGG Table 4 compares the performance of the full MetaCyc dataset, which primarily contains metabolic pathways, to the performance of the 184 KEGG pathways under the ‘Metabolism’ category as well as the full KEGG dataset containing all 502 pathways. We see that the KEGG metabolic pathways alone have the lowest mean MCC and the highest standard deviation. The full KEGG dataset has the highest mean MCC and lowest standard deviation. However, the difference in mean MCCs between the MetaCyc and full KEGG datasets is 0.002, representing a Cohen’s d of 0.2 which is considered to be a small effect size. Also, the percent mean difference is only 0.24%. Thus, the average performance of models trained on the MetaCyc dataset is comparable to that of models trained on the full KEGG dataset. However, if the comparison is limited to metabolic pathways in KEGG, then MetaCyc represents over a 5.6% improvement in prediction performance. View this table: View inline View popup Download powerpoint Table 4 Comparing the overall MCC of the MetaCyc dataset to past KEGG datasets. Fig 9 compares the distribution of MCC for the full KEGG and MetaCyc datasets. Visually, the variance and median are comparable for the KEGG and MetaCyc datasets, which is quantitatively corroborated in Table 4 . Download figure Open in new tab Fig 9 Violinplot comparing the distribution of the MetaCyc dataset’s MCC to that of KEGG. Fig 10a shows the distribution of pathway size (sum of the number of non-hydrogen atoms across all associated compounds) in an overlapping histogram. Fig 10b shows the corresponding smoothed density plot. Metacyc clearly has a unimodal distribution, while KEGG has a bimodal distribution. KEGG has a higher proportion of larger pathways as indicated by the higher blue mode, while MetaCyc has several times the number of larger pathways, even though its proportion is smaller. MetaCyc also has a higher proportion of pathways with 100 to 1000 pathway size, indicated by the the single orange mode. Download figure Open in new tab Fig 10 Distribution and density plot comparing pathway size between the MetaCyc and KEGG datasets. Discussion While the majority of publications on pathway prediction have used KEGG data, the methods clearly are applicable to data from other knowledgebases. In this work, we demonstrate effective training with the MetaCyc knowledgebase for prediction of MetaCyc pathway involvement to a level comparable to the performance on the KEGG knowledgebase. There is less than a 0.24% difference in performance and the Cohen’s d effect size is only 0.2. Pragmatically, the overall performance is equivalent; however, if the comparison is limited to metabolic pathways in KEGG, then MetaCyc represents over a 5.6% improvement in prediction performance. Given the 10-fold larger size of the MetaCyc dataset, we may have reached asymptotic performance level of metabolic pathways with increasing dataset size. Moreover, the MetaCyc results expand the number of pathway definitions that can be predicted by 8-fold. The MetaCyc dataset also contains roughly 50% more compounds than the KEGG dataset, expanding the diversity of compounds available for model training and testing. Another key difference between the MetaCyc and KEGG datasets is the depth of the pathway hierarchy. KEGG pathways only span 3 levels, while MetaCyc pathways span 8 levels. The wider range of pathway granularity may have utility for biological and biomedical interpretation. In addition, the results of the MetaCyc dataset are consistent with past results from the KEGG dataset pertaining to compound size, pathway size, hierarchy depth, and MCC. The deeper you go into the pathway hierarchy, the more granular the pathways become, having fewer associated compounds. The lower number of associated compounds results in smaller pathway size. Pathway performance declines with shrinking pathway size and with increasing depth in the pathway hierarchy. Additionally, the variance of pathway performance decreases as pathway size increases, suggesting that larger pathways have more robust prediction performance. We observe similar results with compound size, with the maximum MCC increasing as compound size increases. We recommend that researchers take compound and pathway size into account when using predicted pathway annotations. The results of the MetaCyc dataset are also consistent with past results regarding the inclusion of higher-level pathways. Even if researchers are primarily interested in lower-level pathways, inclusion of higher-level pathways in the training set results in better performance of the lower-level pathways. Thus, we recommend that the full dataset be used to train a model, even if only a subset of the pathways is used in a downstream analysis. Author Contributions Erik D. Huckvale Obtained the data, wrote the code for the data engineering and analysis, finalized the results, and wrote the first draft of the manuscript. Hunter N. B. Moseley Provided funding, worked in a supervisory role, reviewed the results, revised the manuscript, and provided mentorship. Data availability All data and code for reproducing the results of this manuscript are available in the following figshare: https://doi.org/10.6084/m9.figshare.27317163 . Supporting information Download figure Open in new tab Fig S1 The number of pathways within each hierarchy level. View this table: View inline View popup Table S1 Statistics of performance metrics for models trained on the L1+, L2+, and L3+ datasets. Acknowledgements We thank the University of Kentucky Institute for Biomedical Informatics and National Science Foundation Grant Number 1626364 for their support and associated computing resources. Footnotes https://doi.org/10.6084/m9.figshare.27317163 References 1. ↵ Voet D , Voet JG , Pratt CW . Fundamentals of Biochemistry: Life at the Molecular . 5th ed. Wiley; 2016. 2. Berg JM , Tymoczko JL , Gatto GJ , Stryer L . Biochemistry . 9th ed . New York, NY, USA : W. H. Freeman ; 2019 . 3. ↵ Nelson DL , Cox MM . principles of biochemistry . 8th ed . New York, NY, USA : W. H. Freeman ; 2021 . 4. ↵ Kanehisa M , Goto S . KEGG: Kyoto encyclopedia of genes and genomes . Nucleic Acids Res . 2000 Jan 1; 28 ( 1 ): 27 – 30 . OpenUrl CrossRef PubMed Web of Science 5. ↵ Caspi R , Billington R , Keseler IM , Kothari A , Krummenacker M , Midford PE , et al. The MetaCyc database of metabolic pathways and enzymes - a 2019 update . Nucleic Acids Res . 2020 Jan 8; 48 ( D1 ): D445 – 53 . OpenUrl CrossRef PubMed 6. ↵ Huckvale ED , Moseley HNB . A cautionary tale about properly vetting datasets used in supervised learning predicting metabolic pathway involvement . PLoS ONE . 2024 May 2; 19 ( 5 ): e0299583 . OpenUrl CrossRef PubMed 7. ↵ Huckvale ED , Powell CD , Jin H , Moseley HNB . Benchmark dataset for training machine learning models to predict the pathway involvement of metabolites . Metabolites . 2023 Nov 1; 13 ( 11 ). 8. ↵ Huckvale ED , Moseley HNB . Predicting the pathway involvement of metabolites based on combined metabolite and pathway features . Metabolites . 2024 May 7; 14 ( 5 ). 9. ↵ Huckvale ED , Moseley HNB . Predicting the Association of Metabolites with Both Pathway Categories and Individual Pathways . Metabolites . 2024 Sep 21; 14 ( 9 ). 10. ↵ Huckvale ED , Moseley HNB . Predicting the pathway involvement of all pathway and associated compound entries defined in the kyoto encyclopedia of genes and genomes . Metabolites . 2024 Oct 27; 14 ( 11 ): 582 . OpenUrl CrossRef 11. ↵ Dalby A , Nourse JG , Hounshell WD , Gushurst AKI , Grier DL , Leland BA , et al. Description of several chemical structure file formats used by computer programs developed at Molecular Design Limited . J Chem Inf Model . 1992 May 1; 32 ( 3 ): 244 – 55 . OpenUrl CrossRef 12. ↵ Jin H , Moseley HNB . Hierarchical Harmonization of Atom-Resolved Metabolic Reactions across Metabolic Databases . Metabolites . 2021 Jun 30; 11 ( 7 ). 13. Jin H , Mitchell JM , Moseley HNB . Atom Identifiers Generated by a Neighborhood- Specific Graph Coloring Method Enable Compound Harmonization across Metabolic Databases . Metabolites . 2020 Sep 11; 10 ( 9 ). 14. ↵ Jin H , Moseley HNB . md_harmonize: A Python Package for Atom-Level Harmonization of Public Metabolic Databases . Metabolites . 2023 Dec 17; 13 ( 12 ). 15. ↵ Verstraeten G , Van den Poel D. Using Predicted Outcome Stratified Sampling to Reduce the Variability in Predictive Performance of a One-Shot Train-and-Test Split for Individual Customer Predictions . ICDM (Posters ). 2006 ; 214 . 16. ↵ Chicco D , Starovoitov V , Jurman G . The benefits of the matthews correlation coefficient (MCC) over the diagnostic odds ratio (DOR) in binary classification assessment . IEEE Access . 2021 ; 9 : 47112 – 24 . OpenUrl CrossRef 17. ↵ Rossum GV , Drake FL. Python 3 Reference Manual. CreateSpace ; 2009 . 18. ↵ The pandas development team. pandas-dev/pandas: Pandas 1.0.3. Zenodo . 2020 ; 19. ↵ Harris CR , Millman KJ , van der Walt SJ , Gommers R , Virtanen P , Cournapeau D , et al. Array programming with NumPy . Nature . 2020 Sep 16;585(7825):357–62. 20. ↵ Collette A. Python and HDF5. O’Reilly ; 2013 . 21. ↵ Falcon W , Borovec J , Wälchli A , Eggert N , Schock J , Jordan J , et al. PyTorchLightning/pytorch-lightning: 0.7.6 release. Zenodo . 2020 ; 22. ↵ Paszke A , Gross S , Massa F , Lerer A , Bradbury J , Chanan G , et al. PyTorch: An Imperative Style, High-Performance Deep Learning Library . arXiv. 2019 ; 23. ↵ Pedregosa F , Varoquaux G , Gramfort A , Michel V , Thirion B , Grisel O , et al. Scikit- learn: Machine Learning in Python . arXiv . 2012 ; 24. ↵ Chamberlin D. SQL. In: Liu L, Özsu MT, editors. Encyclopedia of database systems . Boston, MA : Springer US ; 2009 . p. 2753 – 60 . 25. ↵ Raasveldt M , Mühleisen H. Duckdb: an embeddable analytical database. Proceedings of the 2019 International Conference on Management of Data. New York, NY, USA: ACM ; 2019 . p. 1981–4. 26. ↵ Kluyver T , Ragan-Kelley B , Pérez F , Granger B , Bussonnier M , Frederic J , et al. Jupyter Notebooks - a publishing format for reproducible computational workflows. In: Loizides F, Scmidt B, editors. Positioning and Power in Academic Publishing: Players, Agents and Agendas . Netherlands : IOS Press ; 2016 . p. 87 – 90 . 27. ↵ Waskom M. seaborn: statistical data visualization. JOSS . 2021 Apr 6;6(60):3021. 28. ↵ Hunter JD . Matplotlib: A 2D Graphics Environment . Comput Sci Eng . 2007 ; 9 ( 3 ): 90 – 5 . OpenUrl CrossRef PubMed 29. ↵ Salesforce. Tableau Public. Salesforce ; 2024. View the discussion thread. Back to top Previous Next Posted November 03, 2024. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Predicting the pathway involvement of metabolites annotated in the MetaCyc knowledgebase Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Predicting the pathway involvement of metabolites annotated in the MetaCyc knowledgebase Erik D. Huckvale , Hunter N.B. Moseley bioRxiv 2024.10.29.620954; doi: https://doi.org/10.1101/2024.10.29.620954 Share This Article: Copy Citation Tools Predicting the pathway involvement of metabolites annotated in the MetaCyc knowledgebase Erik D. Huckvale , Hunter N.B. Moseley bioRxiv 2024.10.29.620954; doi: https://doi.org/10.1101/2024.10.29.620954 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Systems Biology Subject Areas All Articles Animal Behavior and Cognition (7651) Biochemistry (17746) Bioengineering (13928) Bioinformatics (42066) Biophysics (21499) Cancer Biology (18650) Cell Biology (25579) Clinical Trials (138) Developmental Biology (13409) Ecology (19947) Epidemiology (2067) Evolutionary Biology (24374) Genetics (15633) Genomics (22557) Immunology (17775) Microbiology (40505) Molecular Biology (17217) Neuroscience (88796) Paleontology (667) Pathology (2845) Pharmacology and Toxicology (4836) Physiology (7664) Plant Biology (15179) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9839) Zoology (2272)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00