DrugOn: A Comprehensive Drug Ontology for Precision Oncology

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Introduction Precision oncology and biomedical cancer research increasingly rely on tools to select optimal drugs targeting specific genetic alterations in cancer. A major challenge for bioinformatic tools supporting drug selection is to standardize different annotations (substance, drug name, drug class) to a common level, typically the drug class. While manual classification is time-consuming and potentially biased, existing resources often lack completeness, granularity, or mix up drug classes and drug targets. A structured, automatically built drug ontology such as DrugOn fills these gaps, improving decision support and data-driven research. Methods DrugOn integrates information from multiple sources to create a comprehensive drug ontology. It includes categories, molecular targets and additional annotations from DrugBank, ATC, MesH, KEGG and ClueIO. The combination of this data enables accurate identification of drug categories for each drug. Results DrugOn’s effectiveness was demonstrated by classifying 336 drugs from CIViC. It agreed with manually curated database-derived classifications for 268 out of 282 drugs assessed in translational lymphoma research. In 54 cases, classification was not possible due to data gaps. DrugOn provides a REST API and a front-end application for ontology exploration and automated drug queries. Conclusion DrugOn, a unified drug ontology, is derived from public datasets and refined by precise processing rules to ensure a reliable, updatable resource for drug information, in precision medicine. It uniquely categorizes drug classes and target proteins, and its structured format, complemented by an accessible API, allows for easy integration into data driven pipelines. Initially tailored for lymphoma research, DrugOn’s adaptable nature supports broader cancer research applications and potential data source expansions. DrugOn is accessible at https://mtb.bioinf.med.uni-goettingen.de/drugon-web .
Full text 42,287 characters · extracted from preprint-html · click to expand
DrugOn: A Comprehensive Drug Ontology for Precision Oncology | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search DrugOn: A Comprehensive Drug Ontology for Precision Oncology View ORCID Profile Kevin Kornrumpf , Vera Gnaß , Myrine Holm , View ORCID Profile Tim Beißbarth , View ORCID Profile Raphael Koch , View ORCID Profile Jürgen Dönitz doi: https://doi.org/10.1101/2024.09.23.24314201 Kevin Kornrumpf 1 Department of Medical Bioinformatics, University Medical Center Göttingen , Göttingen, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kevin Kornrumpf For correspondence: kevin.kornrumpf{at}bioinf.med.uni-goettingen.de Vera Gnaß 1 Department of Medical Bioinformatics, University Medical Center Göttingen , Göttingen, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Myrine Holm 1 Department of Medical Bioinformatics, University Medical Center Göttingen , Göttingen, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Tim Beißbarth 1 Department of Medical Bioinformatics, University Medical Center Göttingen , Göttingen, Germany 2 Göttingen Comprehensive Cancer Center (G-CCC) , Göttingen, Germany 3 Campus Institute Data Science Section Medical Data Science (MeDaS) , Göttingen, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tim Beißbarth Raphael Koch 4 Dept. of Hematology and Medical Oncology, University Medical Center Göttingen , Göttingen, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Raphael Koch Jürgen Dönitz 1 Department of Medical Bioinformatics, University Medical Center Göttingen , Göttingen, Germany 2 Göttingen Comprehensive Cancer Center (G-CCC) , Göttingen, Germany 3 Campus Institute Data Science Section Medical Data Science (MeDaS) , Göttingen, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jürgen Dönitz Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Introduction Precision oncology and biomedical cancer research increasingly rely on tools to select optimal drugs targeting specific genetic alterations in cancer. A major challenge for bioinformatic tools supporting drug selection is to standardize different annotations (substance, drug name, drug class) to a common level, typically the drug class. While manual classification is time-consuming and potentially biased, existing resources often lack completeness, granularity, or mix up drug classes and drug targets. A structured, automatically built drug ontology such as DrugOn fills these gaps, improving decision support and data-driven research. Methods DrugOn integrates information from multiple sources to create a comprehensive drug ontology. It includes categories, molecular targets and additional annotations from DrugBank, ATC, MesH, KEGG and ClueIO. The combination of this data enables accurate identification of drug categories for each drug. Results DrugOn’s effectiveness was demonstrated by classifying 336 drugs from CIViC. It agreed with manually curated database-derived classifications for 268 out of 282 drugs assessed in translational lymphoma research. In 54 cases, classification was not possible due to data gaps. DrugOn provides a REST API and a front-end application for ontology exploration and automated drug queries. Conclusion DrugOn, a unified drug ontology, is derived from public datasets and refined by precise processing rules to ensure a reliable, updatable resource for drug information, in precision medicine. It uniquely categorizes drug classes and target proteins, and its structured format, complemented by an accessible API, allows for easy integration into data driven pipelines. Initially tailored for lymphoma research, DrugOn’s adaptable nature supports broader cancer research applications and potential data source expansions. DrugOn is accessible at https://mtb.bioinf.med.uni-goettingen.de/drugon-web . Introduction Precision medicine represents a shift in healthcare that enables personalized treatment options based on specific genetic signatures. This approach relies on a comprehensive analysis genetic alterations, resulting in a larger volume of patient-specific data [ 1 ]. To effectively use this data, clinicians typically query multiple databases and clinical trial repositories. The gathered information about specific drugs span a wide range of formats and levels of detail, including drug names, brand names, active compounds, target pathways, approval numbers, and accession codes. Handling of drug information is important for the preparation of Molecular Tumor Boards (MTBs) on the one hand, and data-driven wet lab projects on the other [ 2 , 3 ]. Both rely on the automatic processing of normalized data and the mapping to the common level of a drug class. To retrieve drug categories, traditional classification methods such as the Anatomical Therapeutic Chemical (ATC) Classification System ( https://atcddd.fhi.no/ )and Medical Subject Headings (MeSH) provide information in a one-dimensional, hierarchical view. However, the hierarchical levels of these resources are often not related to the classes that are used by clinicians. The Kyoto Encyclopedia of Genes and Genomes (KEGG) [ 4 ], ClueIO [ 5 , 6 ] and comprehensive repositories such as DrugBank [ 7 – 12 ] offer a unique perspective on drug categorization. Drug-Bank, in particular, is also rich in detailed information on most of the known drugs and chemical compounds, including their targets and mechanisms of action. With the rapidly evolving field of precision medicine the amount of data to process and the number of approved drugs increased in the last ten years [ 13 , 14 ]. The need for comprehensive and integrative data structures is more relevant than ever. However, the main challenge is to integrate these different sources into a coherent, accessible framework. Ontologies are the state of the art method to collect and organize a complex resource like drug information. They consist of ontology classes representing concepts of the domain and a collection of named connections to set them in relation. The resulting graph represents the complex connections between the included entities. They represent an ideal data structure, as they create a standardized language and framework for categorizing and linking information. This enables efficient integration and comparison of drug data from various sources, by organizing related information, such as brand names, mechanisms of action and targets. The need for a unified drug ontology in precision medicine is underscored in data-driven projects, where drugs must be accurately assigned to their respective classes, mechanisms of action, and targets. Current practices often involve manual annotation of this information, a process that is not only work intensive and time-consuming, but also prone to errors and inconsistencies [ 15 ]. Moreover, the deterministic nature of this task is often questioned, given the subjective biases that can influence the decision-making process [ 16 – 18 ]. Already existing ontologies such as the Drug Repurposing Ontology (DRON) [ 19 ], the Systematized Nomenclature of Medicine (SNOMED) [ 20 ], and RXNorm [ 21 ] provide extensive data. However, they do not always meet the specific needs of precision medicine, which requires a more refined and meaningful approach, as these databases do often not cover all the relevant levels and connections, limiting their value in other applications. To fill this gap, we are introducing DrugOn, a comprehensive drug ontology. DrugOn is designed to integrate and normalize the variety of data available from multiple sources to provide a multidimensional hierarchical view of drugs. Unlike static databases, DrugOn is an easily extensible and maintainable framework. With automatic updates it keeps pace with the rapid advances in drug research and knowledge. DrugOn also enhances usability with an interactive graphical interface. This visual exploration helps identify complex associations and patterns, making DrugOn an helpful tool for researchers, clinicians, and data scientists. The ontology is built primarily using the Web Ontology Language (OWL) and integrates several data resources. OWL is an open, well known data-format and facilitates a rich and complex ontology structure for effective representation of drug information. In summary, the development of DrugOn is an important building block in applications supporting precision oncology. By addressing the limitations of existing systems and introducing features tailored to the needs of precision medicine, DrugOn sets a new standard for drug classification, drug repurposing and identifying potential alternative treatment strategies in precision medicine. Its dynamic, integrative and user-friendly design ensures that it will be a valuable resource in personalized medicine. Materials and methods Data Resources The data provided by DrugOn is primarily derived from a variety of established data sources, ensuring comprehensive coverage and reliability. These sources include the Anatomical Therapeutic Chemical (ATC) Classification System, the Medical Subject Headings (MeSH) database, the Kyoto Encyclopedia of Genes and Genomes (KEGG), with a particular focus on drug groups and classes, the ClueIO (version 2018-09-07) database, and the DrugBank database (version 5.1.12, 2024-03-14) [ 4 – 12 ]. These platforms provide a wide range of information, from drug classifications to molecular details, contributing to the scope and power of the ontology. For validation, we conducted a comprehensive analysis of all therapeutic cancer drugs listed in the Clinical Interpretation of Variants in Cancer (CIViC) database [ 22 , 23 ]. Each drug has been annotated with insights from clinical expertise and a comprehensive literature search, and has been specifically assigned to the appropriate drug category in the clinical context (see Table 1 ). View this table: View inline View popup Download powerpoint Table 1. Excerpt of manual drug classification The table shows a selection of drugs annotated with the categories and with DrugOn’s. The first 15 entries show matching annotations for the manual and DrugOn classification, while the last 5 entries show differences due to incorrect interpretations or missing background information on the part of DrugOn. General selection of drug categories The drug classification methodology utilized by DrugOn is a complex, multi-step approach that is designed to ensure the accuracy of drug categorization based on a variety of criteria. The complete workflow is illustrated in Figure 1 . Download figure Open in new tab Figure 1. Automatic selection of general drug classes The flowchart shows the steps DrugOn takes to process an entered drug to find the best match with a drug category. The process initiates with the normalization of the input drug names to their generic equivalents. This includes the examination of the input name to determine whether it is a synonym or an identifier and the subsequent mapping of the name to the appropriate generic drug name. In the rare instance that this assignment is not feasible, the drug name is processed in a subsequent step for further processing if no initial categorization can be made. Once normalization has been completed, a comparison of the drug name in question with other categorization databases is undertaken. The initial database queried is ClueIO, which was selected on the basis of its high degree of similarity to the manual classification. Should the drug in question be identified, the corresponding drug class is then retrieved directly. In the event that no classification is available from ClueIO, the search is expanded first to include KEGG Drug Class and KEGG Groups and continues with the suggested targets and mechanisms of action from DrugBank. The categories that correspond to the specified combination of target and mechanism (e.g., “PARP1” + “inhibition,” resulting in “PARP inhibitor” if it is available in other resources) are selected as the drug class. Ultimately, the accessibility of the drug name within the Medical Subject Headings (MeSH) or Anatomical Therapeutic Classification (ATC) databases is evaluated. Subsequently, these sources are consulted, as they provide a more expansive range of categories and a more comprehensive description of the drug. In instances where the primary data sources are unable to provide precise information regarding a specific drug name, an alternative approach to the classification process is employed. This involves dividing the entered drug name into its individual substrings (e.g., “Selumetinib in BRAF-mutated tumors” becomes [“Selumetinib”, “in”, “BRAF”, “mutated”, “tumors”]). This method allows for the repetition of the preceding classification steps for each of the resulting parts. In the absence of a precise match in the initial stages, the subsequent phase searches for partial matches across all available categories. The Levenstein distance is utilised to identify the most suitable matches, which are then filtered using a manually defined threshold. In the case of a high number of matches exceeding three, no category is assigned. In contrast, when three or fewer matches are identified, these are presented as probable drug categories. Software and Implementation The ontology was created with the Python 3.9.7 programming language and the Owlready2 package (version 0.38), which is specifically designed for ontologyoriented programming. The software facilitates the creation of OWL ontologies in the form of classes and allows the inclusion of properties for the purpose of storing class-relevant information. An application programming interface (API) has been developed to facilitate access to data from the ontology. The provided API utilizes the Flask package (version 2.1.2), a lightweight and robust web framework for Python, and incorporates a SwaggerUI documentation to enhance usability and accessibility. The documentation is available at https://mtb.bioinf.med.uni-goettingen.de/drugon/v1/doc/ . The front-end application has been implemented using Angular (version 16.1.4). Furthermore, the ontology-based answer (OBA) [ 24 ] server and Onto-Scope are incorporated for the graphical visualization of the ontology. It was designed with an intuitive interface to allow users to efficiently navigate and interact with the rich data stored in DrugOn. The application is accessible at the following URL: https://mtb.bioinf.med.uni-goettingen.de/drugon-web/ and the project’s source code is available at the following GitLab link: https://gitlab.gwdg.de/MedBioinf/mtb/drugon . Results Ontology Architecture Ontology Classes The structure of our recently developed DrugOn ontology hierarchy is based on three basic classes: Drug, Category, and Target (see Figure 2 ). These classes represent the fundamental elements of the DrugOn architecture, with each class fulfilling a distinct role in the organization of drugrelated information. The entities of the domain are added as children to the ontology based on their respective basic class. All drugs are direct children of the base class drugs, other basic classes and the membership in drug classes and their targets are modeled with relationships to the children of the other basic classes. Download figure Open in new tab Figure 2. The principal architectural framework of the DrugOn ontology. The figure was exported by the on-tology editor Protégé [ 25 ] and illustrates the basic classes, Category, Target , and Drug , along with their DrugOn relations. The drug classes are inherited from Category provided by the is_a relation. The relations to Drug are named after their originating sources and may be, for example, has_atc, has_clue_io , and so forth. The Genenames are inherited by the Target class. In relation to Drug , they indicate the action on these specific targets and are described as is_inhibitor_for, is_antagonist_for , e.t.c. The drug terms identified on DrugBank are classified as subsets of the Drug class. A comprehensive representation of each drug is provided, including detailed information such as different identifiers (Chembl [ 26 ], ChEBI [ 27 ], KEGG [ 4 ], etc.) and synonyms. The Target basic class is concerned with the interaction between drugs and human proteins. It provides a link between the actions of drugs and biological entities. The MeSH terms, ATC codes, ClueIO annotations, [ 5 , 6 ] and KEGG drug classifications [ 4 ], that describe a particular drug at various levels are children of the Category. Relation Types Relations are a fundamental component of ontologies, defining the functional connections between to classes named Domain and Range. Similar to classes specialization of relationships can be implemented by inheritance, also called sub-relations. They can also be defined as inverse, indicating the reverse relationship. In DrugOn the four basic inverse relation types in DrugOn are targets/is_targeted, categorizes/is_categorizing , each has subtype relations describing the specific relation and reflects the ‘Actions’-annotation in DrugBank. For example, the relation [Drug] is_inhibitor_for [Target], [Drug] is_antagonist_for [Target] or [Drug] has_atc [Category] can be seen in Figure 3 . Download figure Open in new tab Figure 3. Ontoscope analysis in case of Olaparib Ontoscope shows an example for the drug Olaparib. A) By searching for Olaparib, it is possible to expand the relationship ‘is_inhibitor_for’ (orange edges) to show all proteins that can be inhibited by the selected drug. It shows an inhibitory effect on the three targets PARP1, PARP2 and PARP3. By extending the relation ‘has_atc’ (green edges), Ontoscope shows all available categories from the ATC classification. This shows that Olaparib can be defined as an antineoplastic agent and a poly(ADP-ribose) polymerase (PARP) inhibitor. B) Search for alternative therapies with the same drug categories. In this case the selected database is ClueIO, which declares Olaparib as PARP Inhibitor -additionally there are 23 alternative drugs declared with the same category like Rucaparib and Niraparib. The definition of these relationships provides a more detailed and accurate representation of a drug’s relation. In conclusion, DrugOn’s architectural framework, with its hierarchical structure and comprehensive class relationships, provides a sophisticated platform for representing and analyzing drug information. Its capacity to encapsulate nuanced drug data, from biological targets to categorizations, renders it a valuable resource in precision medicine research. Implementation of the DrugOn Framework The DrugOn framework consists of two main parts. The first one is constructing the ontology, based on Drugbank and enrich the classes with further information from ATC, MesH or ClueIO. Second the search algorithm to query the ontology. A significant challenge arises from the ability to input free text as therapies; this permits the use of non-generic drug names. For instance, we encounter in the therapy column ‘BRAF inhibitors In BRAF Mutant Tumors’, which is clearly understandable to humans, but it is challenging to automatically categorize this within an appropriate framework. DrugOn addresses this issue, by stepwise adjustment of the unmatched input (see Methods). The DrugOn search can be accessed in different ways, depending on the desired scenario. Scenario 1 -General Drug Information One of the primary feature of DrugOn is its web interface, which has been designed with the objective of facilitating user-friendly exploring of the ontology. The search for a drug offers access to all detailed information available in DrugOn. These include the automatic generated drug class, the drug’s mechanism of action, synonyms and identifiers, and all classifications made by other resources. This ease of use is of critical importance to researchers and clinicians who require rapid access to reliable and comprehensive drug information. Consequently, DrugOn represents a resource in both clinical and research settings, providing a readily accessible platform for obtaining essential drug information. Scenario 2 -Visual Ontology exploration In addition to its search functionality, DrugOn offers an advanced interactive ontology viewer. This enables users to search for specific drugs, targets, and categories within the ontology, as well as to explore the complex relationships between these entities, thereby facilitating a deeper understanding of the relationships between various drugs, targets, and categories. One of the key capabilities of this feature is the ability to expand queries to include related nodes. By expanding the ontology by a selected relation type the user can walk through the ontology according to a specific question. This advanced query functionality is particularly valuable for understanding complex drug-target combinations and the nuances of categorization within the ontology. For researchers and clinicians, this provides a comprehensive understanding of the interactions between different drugs and their targets, as well as the classification of drugs within the ontology. This enables a deeper understanding of the mechanisms of action, potential adverse effects, and therapeutic options of various pharmaceutical agents. The capabilities of DrugOn significantly enhance its utility as a tool for both clinical and research applications. Scenario 3 -API and OWL-File for automatic data analysis In order to facilitate large-scale data analysis, Drug-On includes an API that provides automated access to its classification functions as in the other described use cases. This API is crucial for integrating Drug-On’s functionalities into broader data processing workflows. Researchers and clinicians can utilize the API to standardize drug names, perform synonym or identifier queries, and extract information about drug classes, targets, or other relevant attributes. The API supports various endpoints that streamline interactions with the ontology, enabling efficient data handling and analysis. For example, users can query the API to retrieve all drugs associated with a particular mechanism of action or to assign drugs according to their corresponding class. This automation reduces the manual effort required for data processing and ensures consistency across large datasets. DrugOn is implemented using the Web Ontology Language (OWL), a formal specification that facilitates comprehensive and machine-readable representations of knowledge. The OWL ontology file serves as the fourth access path to DrugOn. It can be used for other tools or in environments with restricted internet access. In summary, DrugOn offers a comprehensive range of features, including the provision of detailed drug information through a user-friendly web interface and the possibility of undertaking detailed analysis through its API or locally usable OWL ontology file. Together, these features enhance DrugOn’s applicability in a variety of research and clinical scenarios, making it a valuable tool in the field of drug analysis. Comparison: DrugOn vs manual annotation DrugOn was tested on 336 drugs applied in translational lymphoma research, spanning all available therapy types in CIViC. A comparison of the automatic classification with a manual, database-derived annotation reveals a high degree of correspondence and provides more detailed supplementary information. 54 drugs could not be assigned a classification because no information was available in the underlying data resources or inconsistent manual text entries in the input data. These 17 out of 54 entries failed due to descriptions that were difficult to interpret by machine. For example, descriptions such as ‘corticocorticosteroids In Early Setting’ or ‘MEK inhibitors In BRAF Mutant Tumors’ appear in CIViC, which are clearly understandable for humans, but present a challenge for machine interpretation. In the remaining 282 drug descriptions, the automatic classification returned correct descriptions or matched the manual descriptions. An excerpt of the classifications can be found in Table 1 . The complete table is accessible via the supplementary data and contains the name of each drug, a manual annotation, DrugOn’s classifications, and the drug’s target proteins, including the mechanism of action. DrugOn’s robust performance in accurately classifying a diverse array of CIViC therapies highlights its capabilities. This is not limited to the current version of CIViC but should be also continued with newer versions of the used primary resource, and for other data catalogues, similar to CIViC. However, this functionality represents merely one facet of DrugOn’s utility. It offers a multitude of additional features that address a spectrum of use cases, thereby further enhancing its pivotal role in drug information management. Discussion In the evolving landscape of precision medicine and drug research, we developed DrugOn as a crucial advancement. It demonstrates the power of integrating different databases to create a multidimensional ontology that combines the best features of each resource into a single repository. This innovative approach not only simplifies drug classification, but also improves the accuracy and applicability of drug data in various biomedical projects. The graph based nature of ontologies allows to select the appropriate relation type according to the use case like molecular tumour boards or research projects. Similar DrugOn provides different access options for the use cases such as web based research for manual preparation of the MTB or the API for bioinformatic supported projects. The automatic construction of DrugOn facilitates an up-to-date integration of the included resources. By providing easy access to data on drug classes, targets and similar drugs, DrugOn significantly simplifies the decisionmaking process for clinicians. In addition, DrugOn’s unbiased methodology ensures that the information provided is not skewed by individual experience or biases towards the desired effect, thereby increasing the objectivity and reliability of drug classifications. The use of DrugOn in projects such as the mapping of drug classes in lymphoma data projects and its integration into Onkopus, a framework for cancer variant interpretation in the scope of a MTB ( https://mtb.bioinf.med.uni-goettingen.de/onkopus/ ), exemplifies its versatility and importance. In these applications, DrugOn’s ability to automatically categorize treatments and present comparable options allows for a more nuanced and informed approach to support therapeutic strategies. Existing drug hierarchies such as the ATC ( https://atcddd.fhi.no/ ) and DRON [ 19 ] cover different ranges of hierarchies. For example, ATC is well suited for basic levels, while DRON focuses on drug application and pharmaceutical details. In DrugOn, we integrate multiple data sources to create a comprehensive drug information retrieval tool. Compared to similar approaches, DrugOn has both advantages and limitations. One of its unique features is an ontology-based data source that supports multiple scenarios and allows visual inspection of therapeutic options. Other resources, such as the NCI-Thesaurus ( https://ncit.nci.nih.gov/ncitbrowser/ ) and DRON, also use multidimensional hierarchies. However, DrugOn emphasises a user-friendly graphical ontology browser designed for interactive exploration of research questions. DrugOn also addresses the needs of different research areas by offering different access options. In translational research projects, users can start with the web interface in the early stages of research and later switch to the API for large-scale data processing. Other drug classification systems, such as ATC and DrugBank [ 7 ], provide broad coverage of drugs across all areas of medicine. However, DrugOn’s primary focus is on precision oncology and it includes all DrugBank drugs in its database. In particular, DrugOn offers advanced features that highlight molecular interactions of chemotherapeutic agents. The automatic generation of DrugOn’s ontology has highlighted the need for high quality input data. For example, when evaluating the CIViC [ 23 ] data mappings, a significant number of drugs could not be assigned because the free text annotations did not contain the correct drug names. Another class of unmatched drugs are those for which there is insufficient data in DrugBank. Future versions of DrugBank are expected to address these gaps. However, the field of oncology is rapidly evolving and new drugs are constantly being developed [ 13 ]. DrugOn is a detailed and easy-to-use platform that significantly improves the ability to classify and use drug data and explore alternative treatment options. In short, DrugOn improves the way we use drug information in precision medicine. DrugOn’s success in projects such as clinical decision support shows its potential to contribute to healthcare by making information more accessible and reliable. Data Availability All data produced are available online at: https://gitlab.gwdg.de/MedBioinf/mtb/drugon https://gitlab.gwdg.de/MedBioinf/mtb/drugon Author Contributions Idea and Design: KK, JD, RK, TB. Backend implementation: KK, VG, MH. Frontend implementation: KK. Data annotation: RK. Data validation: KK. Paper writing: KK JD; All authors approved the manuscript in the submitted version and take responsibility for the scientific integrity of the work. Funding Volkswagen Foundation [11-76251-12-1 / 19]; Gemeinsame Bundesausschuss [01NVF20006]; Deutsche Krebshilfe [70113602, 70114018]. Acknowledgement We thank the Göttingen Promotionskolleg für Medizinstudierende for the support of Myrine Holm. References [1]. ↵ Joaquin Mateo et al. “Delivering precision oncology to patients with cancer”. en . In: Nat. Med . 28 . 4 ( Apr . 2022 ), pp. 658 – 665 . OpenUrl [2]. ↵ Claudio Luchini et al. “Molecular tumor boards in clinical practice”. en . In: Trends Cancer 6 . 9 ( Sept . 2020 ), pp. 738 – 744 . OpenUrl [3]. ↵ Brenno Pastò et al. “Unlocking the potential of Molecular Tumor Boards: from cuttingedge data interpretation to innovative clinical pathways”. en . In: Crit. Rev. Oncol. Hematol . 199 . 104379 ( July 2024 ), p. 104379 . OpenUrl [4]. ↵ M Kanehisa and S Goto . “KEGG: kyoto encyclopedia of genes and genomes”. en . In: Nucleic Acids Res . 28 . 1 ( Jan . 2000 ), pp. 27 – 30 . OpenUrl [5]. ↵ Aravind Subramanian et al. “A next generation Connectivity Map: L1000 platform and the first 1,000,000 profiles”. en . In: Cell 171 . 6 ( Nov . 2017 ), 1437 – 1452.e17 . OpenUrl [6]. ↵ Steven M Corsello et al. “The Drug Repurposing Hub: a next-generation drug library and information resource”. en . In: Nat. Med . 23 . 4 ( Apr . 2017 ), pp. 405 – 408 . OpenUrl [7]. ↵ Craig Knox et al. “DrugBank 6.0: The Drug-Bank knowledgebase for 2024”. en . In: Nucleic Acids Res . 52 . D1 ( Jan . 2024 ), pp. D1265 – D1275 . OpenUrl [8]. David S Wishart et al. “DrugBank 5.0: a major update to the DrugBank database for 2018”. en . In: Nucleic Acids Res . 46 . D1 ( Jan . 2018 ), pp. D1074 – D1082 . OpenUrl [9]. Vivian Law et al. “DrugBank 4.0: shedding new light on drug metabolism”. en . In: Nucleic Acids Res . 42 . Database issue ( Jan . 2014 ), pp. D1091 – 7 . OpenUrl [10]. Craig Knox et al. “DrugBank 3.0: a comprehensive resource for ‘omics’ research on drugs”. en . In: Nucleic Acids Res . 39 . Database issue ( Jan . 2011 ), pp. D1035 – 41 . OpenUrl [11]. David S Wishart et al. “DrugBank: a knowledgebase for drugs, drug actions and drug targets”. en . In: Nucleic Acids Res . 36 . Database issue ( Jan . 2008 ), pp. D901 – 6 . OpenUrl [12]. ↵ David S Wishart et al. “DrugBank: a comprehensive resource for in silico drug discovery and exploration”. en . In: Nucleic Acids Res . 34 . Database issue ( Jan . 2006 ), pp. D668 – 72 . OpenUrl [13]. ↵ Beatriz G de la Torre and Fernando Albericio . “The pharmaceutical industry in 2023: An analysis of FDA drug approvals from the perspective of molecules”. en . In: Molecules 29 . 3 ( Jan . 2024 ). [14]. ↵ Wolfgang Sadee et al. “Pharmacogenomics: Driving personalized medicine”. en . In: Pharmacol. Rev . 75 . 4 ( July 2023 ), pp. 789 – 814 . OpenUrl [15]. ↵ David Tamborero et al. “The Molecular Tumor Board Portal supports clinical decisions and automated reporting for precision oncology”. en . In: Nat. Cancer 3 . 2 ( Feb . 2022 ), pp. 251 – 261 . OpenUrl [16]. ↵ Kuniko Sunami et al. “A learning program for treatment recommendations by molecular tumor boards and artificial intelligence”. en . In: JAMA Oncol . 10 . 1 ( Jan . 2024 ), pp. 95 – 102 . OpenUrl [17]. Yoichi Naito et al. “Concordance between recommendations from multidisciplinary molecular tumor boards and central consensus for cancer treatment in Japan”. en . In: JAMA Netw. Open 5 . 12 ( Dec . 2022 ), e2245081 . OpenUrl [18]. ↵ Jacqueline K Perez et al. “Concordance in molecular tumor board case reviews in the ASCO TAPUR study”. en . In: JCO Precis. Oncol . 8 ( Mar . 2024 ), e2300615 . OpenUrl [19]. ↵ William R Hogan et al. “Therapeutic indications and other use-case-driven updates in the drug ontology: anti-malarials, antihypertensives, opioid analgesics, and a large term request”. en . In: J. Biomed. Semantics 8 . 1 ( Mar . 2017 ), p. 10 . OpenUrl [20]. ↵ Shaker El-Sappagh et al. “SNOMED CT standard ontology based on the ontology for general medical science”. en . In: BMC Med. Inform. Decis. Mak . 18 . 1 ( Aug . 2018 ), p. 76 . OpenUrl [21]. ↵ Stuart J Nelson et al. “Normalized names for clinical drugs: RxNorm at 6 years”. en . In: J. Am. Med. Inform. Assoc . 18 . 4 ( July 2011 ), pp. 441 – 448 . OpenUrl [22]. ↵ Benjamin M Good et al. “Organizing knowledge to enable personalization of medicine in cancer”. en . In: Genome Biol . 15 . 8 ( Aug . 2014 ), p. 438 . OpenUrl [23]. ↵ Malachi Griffith et al. “CIViC is a community knowledgebase for expert crowdsourcing the clinical interpretation of variants in cancer”. en . In: Nat. Genet . 49 . 2 ( Jan . 2017 ), pp. 170 – 174 . OpenUrl [24]. ↵ Jürgen Dönitz and Edgar Wingender . “The ontology-based answers (OBA) service: a connector for embedded usage of ontologies in applications”. en . In: Front. Genet . 3 ( Oct . 2012 ), p. 197 . OpenUrl [25]. ↵ Mark A Musen and Protégé Team . “The protégé project: A look back and a look forward”. en . In: AI Matters 1 . 4 ( June 2015 ), pp. 4 – 12 . OpenUrl [26]. ↵ Barbara Zdrazil et al. “The ChEMBL Database in 2023: a drug discovery platform spanning multiple bioactivity data types and time periods”. en . In: Nucleic Acids Res . 52 . D1 ( Jan . 2024 ), pp. D1180 – D1192 . OpenUrl [27]. ↵ Janna Hastings et al. “ChEBI in 2016: Improved services and an expanding collection of metabolites”. en . In: Nucleic Acids Res . 44 . D1 ( Jan . 2016 ), pp. D1214 – 9 . OpenUrl View the discussion thread. Back to top Previous Next Posted September 24, 2024. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following DrugOn: A Comprehensive Drug Ontology for Precision Oncology Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share DrugOn: A Comprehensive Drug Ontology for Precision Oncology Kevin Kornrumpf , Vera Gnaß , Myrine Holm , Tim Beißbarth , Raphael Koch , Jürgen Dönitz medRxiv 2024.09.23.24314201; doi: https://doi.org/10.1101/2024.09.23.24314201 Share This Article: Copy Citation Tools DrugOn: A Comprehensive Drug Ontology for Precision Oncology Kevin Kornrumpf , Vera Gnaß , Myrine Holm , Tim Beißbarth , Raphael Koch , Jürgen Dönitz medRxiv 2024.09.23.24314201; doi: https://doi.org/10.1101/2024.09.23.24314201 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Pharmacology and Therapeutics Subject Areas All Articles Addiction Medicine (574) Allergy and Immunology (865) Anesthesia (304) Cardiovascular Medicine (4460) Dentistry and Oral Medicine (445) Dermatology (383) Emergency Medicine (611) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1517) Epidemiology (15251) Forensic Medicine (31) Gastroenterology (1132) Genetic and Genomic Medicine (6621) Geriatric Medicine (669) Health Economics (1002) Health Informatics (4564) Health Policy (1372) Health Systems and Quality Improvement (1617) Hematology (544) HIV/AIDS (1272) Infectious Diseases (except HIV/AIDS) (15938) Intensive Care and Critical Care Medicine (1107) Medical Education (624) Medical Ethics (147) Nephrology (670) Neurology (6642) Nursing (346) Nutrition (1001) Obstetrics and Gynecology (1148) Occupational and Environmental Health (957) Oncology (3350) Ophthalmology (981) Orthopedics (369) Otolaryngology (421) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1698) Pharmacology and Therapeutics (694) Primary Care Research (714) Psychiatry and Clinical Psychology (5464) Public and Global Health (9259) Radiology and Imaging (2212) Rehabilitation Medicine and Physical Therapy (1372) Respiratory Medicine (1198) Rheumatology (598) Sexual and Reproductive Health (716) Sports Medicine (533) Surgery (715) Toxicology (100) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a039bb366bafdf94',t:'MTc4MDEwMjA4Ng=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00