Full text
40,365 characters
· extracted from
preprint-html
· click to expand
Deep Modeling of Gain-of-Function Mutations on Androgen Receptor | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Deep Modeling of Gain-of-Function Mutations on Androgen Receptor Jiaying You , Jane Foo , Nada Lallous , Artem Cherkasov doi: https://doi.org/10.1101/2025.01.20.633961 Jiaying You 1 Department of Urologic Sciences, Faculty of Medicine, Vancouver Prostate Centre, University of British Columbia Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jane Foo 1 Department of Urologic Sciences, Faculty of Medicine, Vancouver Prostate Centre, University of British Columbia Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nada Lallous 1 Department of Urologic Sciences, Faculty of Medicine, Vancouver Prostate Centre, University of British Columbia Find this author on Google Scholar Find this author on PubMed Search for this author on this site Artem Cherkasov 1 Department of Urologic Sciences, Faculty of Medicine, Vancouver Prostate Centre, University of British Columbia Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: acherkasov{at}prostatecentre.com Abstract Full Text Info/History Metrics Preview PDF Abstract The efficiency of Androgen Receptor (AR) pathway inhibitors for prostate cancer (PCa) is on decline due to resistance mechanisms including the occurrence of gain-of-function mutations on human androgen receptor (AR). Hence, understanding and predicting such mutations is crucial for developing effective PCa treatment strategies. Leveraging accumulated data on clinically relevant AR mutants with recent advances in deep modeling techniques, this study aims to unveil and quantify critical AR mutation-drug relationships. By incorporating molecular descriptors for drugs and mutated genes sequences, this work represented these features as single vectors and demonstrates their effectiveness in modeling AR mutant responses to conventional antiandrogens. The developed approach achieves up to 80% accuracy in predicting the gain-of-function behavior of AR mutants and therefore can potentially uncover unknown agonist/antagonist relationships among mutant-drug pairs. 1 Introduction Prostate cancer is the second most common cancer in men, exerting a significant impact on their longevity and quality of life [ 1 ]. The traditional treatment is by inhibiting the androgen receptor (AR) activity, which regulates biological responses in males, by inhibiting androgen production or interfering with their binding to the AR. However, confidence in AR-targeted drugs like enzalutamide [ 2 ] and abiraterone [ 3 ] is fading, as more patients exhibit adaptive resistance to these therapies. Structurally, the AR gene is located on the X chromosome [ 4 ] and encodes three major domains: (1) the N terminal domain (NTD, residues 1-559), (2) the DNA binding domain (DBD, residues from 560-625), and (3) the C-terminal ligand binding domain (LBD, residues from 671 to 920). A flexible hinge region links the DBD to the LBD. Over the years, there has been tremendous progress in developing AR inhibitors with novel modalities [ 5 ] [ 6 ] [ 7 ]. However the LBD remains the only domain targeted by clinically approved antiandrogens such as darolutamide[ 8 ] and apalutamide[ 9 ], despite its susceptibility to resistance and gain-of-function mutations. To address this challenge, our lab has developed small molecules that disrupt AR function by targeting either the NTD [ 10 ] or the DBD [ 11 ]. Additionally, we are developing prediction models to assess the response of AR point mutation to inhibition by LBD-directed agents, employing machine learning techniques to capture agonist and antagonist behaviors within mutant-drug pairs. In recent years, Quantitative Structure-Activity Relationship (QSAR) modeling has become as an effective method in bioinformatics, enabling researchers to elucidate the biological activity of chemical compounds based on their biomarkers such as molecular structures and protein sequences [ 12 ]. QSAR models can support various applications such as drug discovery [ 13 ] and toxicology prediction [ 14 ], by analyzing the interactions between biomarker descriptors and biological responses. The adoption of deep learning (DL)[ 15 ] models over traditionally used linear regression or statistical methods has significantly improved performance in this field, enabling the capture of complex patterns within high-dimensional data. DL models excel in automatically extracting informative embeddings from raw data inputs and learning through tailored iterations, making them well-suited for addressing many bioinformatics challenges, where biological datasets are often massive and biased. By leveraging neural networks, DL enhances the precision of QSAR models, allowing for higher accuracy in predicting compound activity, and empowering the exploration of novel molecular designs in large chemical spaces. To provide a comprehensive understanding of how AR mutations impact responses to anti-androgen therapies and to foresee potential therapeutic strategies for patients with resistant prostate cancer, we incorporated data from previous research by Lallous et al. [ 16 ], which investigated mutant-drug interactions for 86 AR mutants across a panel of five anti-androgen drugs. Using this combined dataset, we successfully predicted AR mutant-drug interactions through machine learning technologies, including DL and traditional bench-marks. Using this combined dataset, we managed to predict AR mutant-drug interactions using machine learning technologies including DL and traditional benchmarks. Our experiment involved 4 major components: (1) dataset outlier detection, (2) feature generation and selection, (3) cross-validation, and (4) benchmark comparison. Through these steps, we established a robust classification model for AR mutant-drug pairs. This framework leverages machine learning techniques to model the complex relationships between AR mutations and their biological responses. Our in-house dataset was generated from multiple experiments on 3 chemicals: enzalutamide, bicalutamide and darolutamide with 12 common mutants ( Figure 1 ), where each AR-mutant pair was tested 8 times under 7 doses, were recorded the luciferase activities among all experiments and excluded values as outliers with 10% percentile (below the bottom 10% or above the top 90% percentile). Download figure Open in new tab Figure 1: The effects of major antiandrogens were evaluated on 12 clinically observed resistant mutants. The graph features mutants’ responses to enzalutamide, bicalutamide and darolutamide. WT–the wild type, nonmutated AR. Red blocks are examples being labeled for up-(F877L_T878A), down-(M896T), mixed-(V716M) trend on enzalutamide. Feature generation is also a critical step in machine learning modeling, here we utilized a variety of parameters including QSAR descriptors for drugs, and amino acid compositions for proteins, combined with physicochemical properties, to create a rich representation of mutant-drug compounds pairs. To retrieve robust prediction results, we used k-fold cross validation (k=3) when applying the DL model and our benchmark models including Support Vector Machine (SVM) [ 17 ] and Random Forest (RF) [ 18 ] , the model metric was then compared, we observed our DL model out-performs all traditional machine learning models. In this work, we present a DL model designed to classify mutant-drug pairs based on their activity profiles. To evaluate its performance, we compared our deep neural network with established machine learning benchmarks, including random forest and support vector machines (SVM). Our aim is to leverage deep learning for multi-class classification tasks, enabling the accurate classification of drug responses in AR mutants. 2 Methods 2.1 Laboratory procedure for conducting a functional assay on PC3 cells 2.1.1 Cell lines PC3 cells were purchased from the American Type Culture Collection (ATCC) and were maintained in RPMI 1640 (Gibco, Life Technologies) supplemented with 10% fetal bovine serum (FBS). The cells were cultured in a humidified incubator at 37°C and 5% CO2 and routinely checked for mycoplasma contamination. 2.1.2 Luciferase transcriptional assay PC3 cells were starved for 72 hours in RPMI 1640 supplemented with charcoal-stripped FBS (CSS) then seeded into a 96-well plate at a density of 5000 cells/well. The following day, the cells were transfected with pcDNA-AR (wildtype or mutant) and ARR3tk-luc luciferase reporter plasmid [ 19 ] 24 hours post transfection, the cells were stimulated with 0.1nM R1881 and treated with a 1:3 serial dilution from 50 µ M to 0.07 µ M of enzalutamide simultaneously. After 24 hours, the cells were lysed with 50 µ l of passive lysis buffer (Promega #E1910) and 20 µ l of lysate was transferred to a white flat-bottom 96-well plate (Corning Life Sciences Cat#3912). Luciferase activity was measured using luminescence readings after adding 50 µ l of luciferase reagent (Promega Luciferase Assay System #E1500). Luciferase activity was normalized and expressed as a percentage of wildtype AR activity. 2.2 Dataset generation In this work we additionally considered 12 clinically relevant AR mutants, which were experimentally tested against 3 known AR antiandrogens: enzalutamide, bicalutamide and darolutamide. We recorded their responses by measuring the luciferase-labeled transcriptional activities. The assay involves transfecting the cells with wild-type or mutated AR, treating them with compounds of interest, and measuring luminescence as an indicator of AR transcription. The results are then normalized and expressed in percentage of wild-type AR activity. We labelled the dose-response reaction amino acid substitutions of AR by common anti-androgen drugs into 3 categories, antagonist agonist and mixed, where we observe a up/mixed/down pattern correspondingly shown in red blocks in Figure 1 on various mutants treated with enzalutamide. To enrich our dataset and enhance the depth of modeling, we augmented the data by incorporating findings from the previous work [ 20 ] on AR mutants. In particular, we integrated data from an experiment investigating the reactions of 86 AR mutants to 5 anti-androgen chemicals: bicalutamide, enzalutamide, hydroxyflutamide, apalutamide and darolutamide. This dataset was originally published by Lallous et al. Combined with the experimental data for 12 additional mutants from this study, we labelled the resulting combined datasets with the following rules: A repeated of eight experiments were conducted at each concentration for a single mutant-drug pair, we first calculated the modified z-score, and removed data points from repeated experiments with outliers, threshold for this step is two standard deviations. Average of luciferase activity scores were then compared on different concentration levels from -2 to 2 (in log form, unit: µ M). If the luciferase activation values only increase with higher concentration, we label such pairs as class 1 (up pattern class). If luciferase activation values only decrease with higher concentration, we label such pairs as class 2 (down pattern class). If we observe a mixed pattern of ups and downs on various concentration values, we label those pairs as class 3(mixed pattern class). Note that if the luciferase activation values are stable with all drug doses, we included those pairs in the mixed pattern class as well. These labeling are later being used in multi-class classification modeling. A combination of the above dataset was generated eventually, including five unique anti-androgen compounds and 86 unique mutants. 238 mutant-drug pairs were labelled for either down, up or mixed (no response). 2.3 Feature generation Feature representations are critical in deep modelling and the common feature generation technologies include one-hot encoding [ 21 ] , image feature extraction [ 22 ] , time series modeling [ 23 ] , data normalization and scaling [ 24 ] . The objective of finding appropriate feature is to transfer data points into the format that can be comprehended to neural networks. In artificial intelligence, feature vectors formed the input layer, the starting point where massive neuros in hidden layers can be learned and progressed actively during the intensive computational steps. Sequence embedding is one of the most common and effective encoding for sequence data point such as genes [ 25 ]. In our study, we focused on mutated sequence data and encoded the mutants with descriptors. Those descriptors include: QSAR descriptors We explored molecular parameters including BLOSUM indices [ 26 ], Cruciani properties [ 25 ] , FASGAI vectors [ 27 ] , Kidera factors [ 28 ] , MS-WHIM scores [ 29 ] , ProtFP descriptors [ 30 ] , ST-scales [ 31 ] , T-scales [ 32 ] , VHSE-scales [ 26 ] , IND descriptors [ 33 ] [ 34 ] [ 35 ] and Z-scales [ 18 ] . Those QSAR descriptors capture various physicochemical and structural properties of molecules. These descriptors can be useful for characterizing the chemical and structural changes induced by mutations and predicting their effects on protein function or interaction with other molecules. Amino acid counts and frequency Analyzing the counts and frequencies of different amino acids in mutant sequences can provide insights into the composition and distribution of amino acids, which could be important for understanding the impact of mutations on protein structure and function. Sequence profiles These included hydrophobicity, hydrophobic moment, and membrane position, provide information about the local environment and structural features of amino acids within a protein sequence. Physicochemical Properties Physicochemical properties like the aliphatic index, instability index, theoretical net charge, isoelectric point, and molecular weight provide additional insights into the biochemical characteristics of mutant proteins. These properties can influence protein stability, solubility, and interaction with other molecules, making them relevant for understanding the functional consequences of mutations. Biological Properties Predicting structural class and other biological properties based on mutant sequences can provide valuable information about the potential functional impact of mutations. This could include predictions of protein secondary structure, subcellular localization, or functional domains. Morgan fingerprints These fingerprints generally used in machine learning models for molecular embeddings in various domain such as drug-drug interaction [ 36 ] prediction, drug-target interaction [ 37 ] prediction and drug repurposing [ 38 ] . Morgan fingerprints is considered a balance between simplicity, efficiency, and representational power, making them one of the most widely used fingerprinting methods in cheminformatics. In this work, we extracted drug descriptors from Simplified Molecular Input Line Entry System (SMILE) with python package RDKit (RDKit: Open-source cheminformatics; http://www.rdkit.org ), and generated morgan fingerprints with radius = 2, meaning with 2 diameters of the atom envi ronment are considered for encoding. We generated 128 bits of morgan fingerprints as drug descriptors embedded with binary 0/1 vectors. We later combined heterogenous encodings of mutants and drugs into single vectors and fed them into the deep neural network for deep modeling and other traditional machine learning benchmark models. 2.4 Deep modeling Deep learning is an artificial intelligence technology that learns data points through a bunch of neurons and layers actively by forward and backward propagations. The weighs assigned to each neural are assigned initially with non-sense numbers as a starting point, and graduate begin to update towards one goal: minimize the difference between real target value and predicted target value by the learned linear or unilinear formula. It has been widely applied into various fields such as image detection [ 39 ] , language recognition [ 40 ] , drug discovery [ 38 ] and recommendation systems [ 41 ] . The power of deep learning is the ability to learn automatically like human being with hierarchical feature representations. Compared with traditional machine learning technologies, deep learning can generally take massive data points and process them in an un-linear pattern, making it effective for complex concepts where intricate correlations may not be picked up by tradition linear models ( Figure 2 ). Download figure Open in new tab Figure 2: Deep neural network structure. Usually contains an input layer with sample representations, hidden layers with neurons that learns weighs in training process, and one output layer for classification responses. In this work, we proposed a deep neural network model composing three hidden layers with 64, 32, 32 neurons respectively for multi-class classification. Heterogenous descriptors on mutants and drugs are combined as the input layer, and the AR activity on mutant-drug pairs were labeled as agonist, antagonist and mixed responses as the final output layer. 2.5 Benchmark modeling To validate the performance of our deep neural network (DNN) model, it is crucial to select benchmark models that are widely acknowledged in the field for comparison. Among these, state-of-the-art machine learning algorithms such as Random Forest and Support Vector Machines (SVM) are particularly noteworthy for their effectiveness in classifying data points into three or more categories. 2.5.1 Random Forest Random forest in an ensemble learning method that binds multiple decision trees for training and votes their outputs for final predictions. Using randomly selected sub features, each tree in the random forest system makes a prediction independently, the randomness enables the random forest to capture various patterns within the data. This method is ideal as our benchmark model for predicting mutants-drug responses, as it classifies multi response types by dividing the feature space into different classes. It is particularly effective handling complex relationships with large variables data points by enhancing robustness and preventing overfitting, given its assembling nature of having numerous for independent predictions. 2.5.2 Support Vector Machines (SVM) SVM is a supervised learning model that classifies datapoints by optimizing a hyperplane (N dimensional space) that maximizes the distance between different classes. Those closest data samples are called support vectors for each class, and the goal is to find the decision boundary that maximize the margin between those support vectors. The decision boundary will be a straight line in binary classification problems if a linear relationship is caught between labels and features, and a higher dimensional hyperplane will be utilized when datapoints are not linearly separable. To handle the non-linearity of datapoints. Here we applied non-linear SVM that utilizes kernel function and brings dataset to a higher dimension, where again a straight line can be achieved separating different classes linearly. In this paper, we measured the performance using Random Forest and SVM on mutant-drug pairs using identical embedded molecular descriptors. By comparing their performance metrics obtained from our deep learning model and benchmark models, we aim to gain insights into the strengths and limitations of each approach, highlighting the advantages of deep learning in this specific application. 3 Results 3.1 Data generation In this study, we constructed the experiment that records the drug responses of various androgen receptor (AR) mutants to anti-androgen drugs enzalutamide, bicalutamide, darolutamide, hydroxyflutamide, apalutamide. Each of the experiment contains unique AR mutants, each tested by a single drug under seven different concentrations. To ensure the integrity and reliability of our dataset, we applied a multi-step labeling process: Outlier Removal Each experimental result detected for outlies before labelling. We identified extreme values in data points that fell outside the 10th and 90th percentiles. Average Calculation After outlier removal, we calculated the average luciferase activities for each mutant-drug pair across the different concentrations. Trend Classification Upward Trend Labeled as “up” if luciferase activity consistently increased with higher concentrations. Downward Trend Labeled as “down” if activity consistently decreased with higher concentrations. Mixed Trend Labeled as “mixed” if the response exhibited both increases and decreases or no response across different concentrations. This systematic approach resulted in a dataset containing nearly 240 mutant-drug pairs, each classified into one of the three categories (up, down, or mixed). We visualized the statistics on interaction pairs on 5 unique drugs ( Figure 3 ) and 86 unique mutants ( Figure 4 ) respectively in below graphs. Download figure Open in new tab Figure 3: Count of mutant-drug response types on 5 antiandrogen drugs: enzalutamide, bicalutamide, darolutamide, hydroxyflutamide, apalutamide in the combined dataset Download figure Open in new tab Figure 4: Count of mutant-drug response types on 65 unique mutants in dataset 3.2 Deep learning model performance We conducted model training through a systematic approach to predict the responses of androgen receptor mutants to anti-androgen drugs. Initially, we merged comprehensive features in one vector including drug embedding and mutant descriptors. The response labels were encoded into numeric classes. Feature dimensionality was reduced using Principal Component Analysis (PCA) [ 42 ] analysis, retaining 64 principal components before feeding into the neural network model. This optimized feature set was subsequently split using shuffle split 3-fold cross-validation, ensuring robust training and evaluation with 70% of the data for training and 30% for testing. A feedforward neural network was conducted using PyTorch with 3 fully connected hidden layers with ReLU [ 43 ] activations that provides non-linear calculations, and a sigmoid output layer that transfers output into possibilities. The model was trained using binary cross-entropy loss and stochastic gradient descent for optimization. Performance metrics including accuracy, precision, recall, and AUC-ROC were monitored to evaluate the model’s effectiveness in below figure 6 . Finally, the trained model’s predictions were compiled for all mutant-drug combinations, enabling the identification of potential new drug candidates for experimental validation. The DNN model attained an accuracy score of 0.86, an AUC values above 0.91 ( Figure 5 ) and the lowest precision score of 0.8. We also plotted the training losses and accuracy over epochs in visual to prevent over fitting or under fitting shown in figure 6 . Download figure Open in new tab Figure 5: DNN model performance in accuracy, AUCand precision. The model accuracy is 0.85 and the AUC values above 0.89 wasachieved Download figure Open in new tab Figure 6: Training loss decreases, and recall/accuracy increases in training iterations. 3.3 Benchmark model performance To evaluate the performance of the predictive models, we employed two widely used machine learning algorithms: Support Vector Machine (SVM) and Random Forest (RF). These models were selected for their robustness in handling multi-class classification tasks and their ability to provide interpretable results. We first implemented Support Vector Machine using Scikit-learn library in python, and enabled probability in returns for AUC calculation. To be consistent with the primary neural work model the SVM was trained on the same integrated dataset consisting of feature representations generated from mutant and drug descriptors. The performance of the SVM model was evaluated an overall accuracy of 83%. The Random Forest model was constructed using the Scikit-learn library again using the same feature set generated from mutant-drug pairs. The RF model’s performance was also evaluated, yielding an overall accuracy of 84%. The performance of benchmark classification models was evaluated using the Area Under the Curve (AUC) of the Receiver Operating Characteristic (ROC) curve to compare with the primary neural network model. Figure 7 below presents the ROC curves for both the Random Forest and Support Vector Machine (SVM), where the false positive rate (FPR) is displayed on the x-axis, and the true positive rate (TPR) on the y-axis. The ROC curve illustrates the trade-off between sensitivity and specificity at different threshold settings. The diagonal dashed line represents the performance of a random classifier as a baseline for comparison. The AUC values for the models were calculated from the ROC curves for both models. The Random Forest model achieved an AUC of approximately 0.87, indicating its strong capability to discriminate between classes. The SVM model displayed an AUC of approximately 0.90, indicating a better prediction power compared with random forest model. These values suggest that both models exhibit valuable predictive performance, with the SVM model marginally outperforming the random forest in terms of area ROC value shown in below figure 7 . Download figure Open in new tab Figure 7: ROC and precision-recall curves for two benchmark models: Random Forest and SVM. 4 Discussion Our research shows that by using DL models with integrated embeddings from mutant and drug descriptors, we can achieve outstanding performance in predicting AR mutants’ responses towards anti-androgen therapies. While people are losing confidence in traditional treatments due to gain-of-function mutations in the AR, machine learning model shows the potential in foreseeing drug responses that can be utilized in novel drug design or distinguishing possible novel mutants. In this paper, we capitalized on the integration of molecular descriptors and gene sequence data as informative representation and fed them in the unified format to the neural network model. By leveraging the comprehensive dataset of AR mutants and a diverse range of anti-androgen compounds, we were able to achieve an 79% accuracy in mutant-drug pair predictions, underscoring the potential of machine learning techniques in this domain. This approach allows for the exploration of previously insufficiently studied interactions and potentially unveiling new therapeutic strategies. The successful classification of mutant-drug pairs into distinct response categories—agonist, antagonist, and mixed— demonstrates the robustness of our model in capturing the various effects on mutations. Benchmark models including random forest and support vector machine were established for comparison. We calculated the accuracy and plotted their ROC curves to demonstrate their prediction power. Tho the random forest and SVM also shows the abilities in distinguishing drug responses in multi-class on AR mutants, they both underperformed compared with the developed neural network model, which exhibits superior capabilities in capturing non-linear patterns and achieving higher accuracy. These findings demonstrate the potential in deep learning for mutant-drug response prediction and its capacity to facilitate novel medicine therapies for prostate cancer patients. Moreover, to ensure the reliability and integrity of this research, the multi-step cleaning was applied in data generation and labeling process in this study. The processed dataset containing a diverse response of nearly 70 AR mutants and 5 known anti-androgen drugs, becomes a significant resource that can support future investigations into AR biology and therapeutic resistance. 5 Conclusion This paper investigated the efficiency of deep learning models in predicting the agonist/antagonist behavior within AR mutant-drug pairs. The approach relied on the use of combined embeddings on sequence- and molecular QSAR descriptors and enabled high accuracy predictions. This study not only demonstrates potentials in predicting AR mutations-drug responses using deep learning but also enables effective exploration of novel antiandrogens with reduced or eliminated ability active AR mutants. It is anticipated that the developed approach will help tackling clinical challenges caused by AR mutations and ultimately improve prostate cancer patient care. 6 Supplementary information Supplementary data are available online at https://github.com/chill-bear/ar_mutants/tree/add_files . References [1]. ↵ A. Barsouk , S. A. Padala , A. Vakiti , A. Mohammed , K. Saginala , K. C. Thandra , P. Rawla , and A. Barsouk , Medical Sciences , 2020 , 8 . [2]. ↵ M. Tucci , C. Zichi , C. Buttigliero , F. Vignani , G. V. Scagliotti , and M. D. Maio , OncoTargets and Therapy , 2018 , 11 , 7353 – 7368 . OpenUrl CrossRef [3]. ↵ C. Buttigliero , M. Tucci , V. Bertaglia , F. Vignani , P. Bironzo , M. D. Maio , and G. V. Scagliotti , Cancer Treatment Reviews , 2015 , 41 , 884 – 892 . OpenUrl CrossRef PubMed [4]. ↵ D. B. Lubahn , D. R. Joseph , P. M. Sullivan , H. F. Willard , F. S. French , and E. M. Wilson , Science , 1988 , 240 , 327 – 330 . OpenUrl Abstract / FREE Full Text [5]. ↵ Q. Yi , Molecular cancer therapeutics , 2023 , 22 , 570 – 582 . OpenUrl CrossRef PubMed [6]. ↵ M. H. Tan , J. Li , H. E. Xu , K. Melcher , and Y. El , Acta Pharmacologica Sinica , 2015 , 36 , 3 – 23 . OpenUrl PubMed [7]. ↵ R. Narayanan , Asian Journal of Urology , 2020 , 7 , 271 – 283 . OpenUrl CrossRef PubMed [8]. ↵ K. Fizazi , New England Journal of Medicine , 2019 , 380 , 1235 – 1246 . OpenUrl CrossRef PubMed [9]. ↵ K. N. Chi , N. Agarwal , A. Bjartell , H. Byung , A. J. Chung , S. P. De , R. Gomes , Given, J. Álvaro, and Soto, New England Journal of Medicine , 2019 , 381 , 13 – 24 . OpenUrl CrossRef PubMed [10]. ↵ F. Ban , E. Leblanc , A. D. Cavga , C.-C. F. Huang , M. R. Flory , F. Zhang , M. E. K. Chang , H. Morin , N. Lallous , K. Singh , M. E. Gleave , H. Mohammed , P. S. Rennie , N. A. Lack , and A. Cherkasov , Cancers , 2021 , 13 . [11]. ↵ M. Radaeva , F. Ban , F. Zhang , E. Leblanc , N. Lallous , P. S. Rennie , M. E. Gleave , and A. Cherkasov , International Journal of Molecular Sciences , 2021 , 22 , 2493 – 2493 . OpenUrl CrossRef PubMed [12]. ↵ A. Cherkasov , E. N. Muratov , D. Fourches , A. Varnek , I. I. Baskin , M. Cronin , J. Dearden , P. Gramatica , Y. C. Martin , R. Todeschini , V. Consonni , V. E. Kuz’min , R. Cramer , R. Benigni , C. Yang , J. Rathman , L. Terfloth , J. Gasteiger , A. Richard , and A. Tropsha , QSAR Modeling: Where Have You Been? Where Are You Going To , 2014 , 57 , 4977–5010. [13]. ↵ A. Tropsha , O. Isayev , A. Varnek , G. Schneider , and A. Cherkasov , NatureReviews Drug Discovery , 2024 , 23 , 141 – 155 . OpenUrl CrossRef PubMed [14]. ↵ T. M. Whitehead , J. Strickland , G. J. Conduit , A. Borrel , D. Mucs , and Baskerville-Abraham, Journal of Chemical Information and Modeling , 2024 , 64 , 2624 – 2636 . OpenUrl CrossRef PubMed [15]. ↵ Y. Lecun , Y. Bengio , and G. Hinton , Nature , 2015 , 521 , 436 – 444 . OpenUrl CrossRef PubMed [16]. ↵ O. Snow , N. Lallous , M. Ester , and A. Cherkasov , International Journal of Molecular Sciences , 2020 , 21 . [17]. ↵ M. A. Hearst , S. T. Dumais , E. Osuna , J. Platt , and B. Scholkopf , 1998 . [18]. ↵ R. Forest , M. Sandberg , L. Eriksson , J. Jonsson , M. Sjöström , and S. Wold , Journal of Insurance Medicine , 1998 , 47 , 31 – 39 . OpenUrl [19]. ↵ R. Snoek , N. Bruchovsky , S. Kasper , R. J. Matusik , M. Gleave , N. Sato ,. Rennie, and P. S, The Prostate , 1998 , 36 , 256 – 263 . OpenUrl CrossRef [20]. ↵ O. Snow , N. Lallous , M. Ester , and A. Cherkasov , International Journal of Molecular Sciences , 2020 , 21 . [21]. ↵ A. C. H. Choong and N. K. Lee , International Conference on Computer and Drone Applications (IConDA) , 2017 , 60 – 65 . [22]. ↵ G. Kumar and P. K. Bhatia , Fourth International Conference on Advanced Computing & Communication Technologies , 2014 , 5 – 12 . [23]. ↵ N. K. Ahmed , A. F. Atiya , N. Gayar , El, and H. El-Shishiny , Econometric Reviews , 2010 , 29 , 594 – 621 . OpenUrl CrossRef [24]. ↵ H. Akay and S. G. Kim , CIRP Annals , 2020 , 69 , 141 – 144 . OpenUrl CrossRef [25]. ↵ G. Cruciani , M. Baroni , E. Carosati , M. Clementi , R. Valigi , S.. P. Clementi , and G. I. Prajapati , Fifth International Conference on Advanced Computing & Commu-nication Technologies , 2004 , 18 , 41 – 47 . OpenUrl [26]. ↵ H. Mei , Z. H. Liao , Y. Zhou , and S. Z. Li , Cold Spring Harbor Protocols , 2005 , 80 , 775 – 786 . OpenUrl [27]. ↵ G. Liang , G. Chen , W. Niu , and Z. Li , Chemical Biology & Drug Design , 2008 , 71 , 345 – 351 . OpenUrl CrossRef PubMed [28]. ↵ A. Kidera , Y. Konishi , M. Oka , T. Ooi , and H. A. Scheraga , Journal of Protein Chemistry , 1985 , 4 , 23 – 55 . OpenUrl CrossRef Web of Science [29]. ↵ A. Zaliani and E. Gancia , Journal of Chemical Information and Computer Sciences , 1999 , 39 , 525 – 533 . OpenUrl CrossRef [30]. ↵ G. J. Westen , R. F. Swier , J. K. Wegner , A. P. Ijzerman , H. W. V. Vlijmen , and A. Bender , Journal of Cheminformatics , 2013 , 5 . [31]. ↵ L. Yang , M. Shu , K. Ma , H. Mei , Y. Jiang , and Z. Li , Amino Acids , 2010 , 38 , 805 – 816 . OpenUrl CrossRef PubMed [32]. ↵ F. Tian , P. Zhou , and Z. Li , Journal of Molecular Structure , 2007 , 830 , 106 – 115 . OpenUrl CrossRef [33]. ↵ A. Cherkasov , Z. Shi , M. Fallahi , and G. L. Hammond , Journal of Medicinal Chemistry , 2005 , 48 , 3203 – 3213 . OpenUrl CrossRef PubMed Web of Science [34]. ↵ A. Cherkasov , International Journal of Molecular Sciences , 2005 , 6 , 63 – 86 . OpenUrl CrossRef [35]. ↵ K. Hilpert , C. D. Fjell , and A. Cherkasov , Peptide-Based Drug Design 2008 , 127 – 159 . [36]. ↵ S. Dere and S. Ayvaz , Healthcare informatics research , 2020 , 26 , 42 – 49 . OpenUrl CrossRef PubMed [37]. ↵ C. Chen , H. Shi , Z. Jiang , A. Salhi , R. Chen , X. Cui , and B. Yu , Computers in Biology and Medicine , 2021 , 136 . [38]. ↵ J. You , M. Hsing , and A. Cherkasov , Small Molecules on Longevity-Associated Genes. Pharmaceuticals , 2021 , 14 . [39]. ↵ R. Chauhan , K. K. Ghanshala , and R. C. Joshi , First International Conference on Secure Cyber Computing and Communication (ICSCCC) , 2018 , 278 – 282 . [40]. ↵ F. Richardson , D. Reynolds , and N. Dehak , IEEE Signal Processing Letters , 2015 , 22 , 1671 – 1675 . OpenUrl CrossRef [41]. ↵ U. Gupta , C. J. Wu , X. Wang , M. Naumov , B. Reagen , D. Brooks , B. Cottel , K. Hazelwood , M. Hempstead , B. Jia , H.-H. S. Lee , A. Malevich , D. Mudigere , M. Smelyanskiy , L. Xiong , and X. Zhang , IEEE International Symposium on High Performance Computer Architecture (HPCA) , 2020 , 488 – 501 . [42]. ↵ H. Abdi and L. J. Williams , WIREs Computational Statistics , 2010 , 2 , 433 – 459 . OpenUrl CrossRef [43]. ↵ A. F. Agarap , arXiv preprint , 2018 , 1803 .08375. View the discussion thread. Back to top Previous Next Posted January 24, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Deep Modeling of Gain-of-Function Mutations on Androgen Receptor Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Deep Modeling of Gain-of-Function Mutations on Androgen Receptor Jiaying You , Jane Foo , Nada Lallous , Artem Cherkasov bioRxiv 2025.01.20.633961; doi: https://doi.org/10.1101/2025.01.20.633961 Share This Article: Copy Citation Tools Deep Modeling of Gain-of-Function Mutations on Androgen Receptor Jiaying You , Jane Foo , Nada Lallous , Artem Cherkasov bioRxiv 2025.01.20.633961; doi: https://doi.org/10.1101/2025.01.20.633961 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7642) Biochemistry (17715) Bioengineering (13907) Bioinformatics (42003) Biophysics (21470) Cancer Biology (18624) Cell Biology (25533) Clinical Trials (138) Developmental Biology (13390) Ecology (19935) Epidemiology (2067) Evolutionary Biology (24356) Genetics (15617) Genomics (22529) Immunology (17753) Microbiology (40432) Molecular Biology (17200) Neuroscience (88681) Paleontology (667) Pathology (2840) Pharmacology and Toxicology (4828) Physiology (7653) Plant Biology (15161) Scientific Communication and Education (2046) Synthetic Biology (4304) Systems Biology (9826) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.