Full text
37,311 characters
· extracted from
preprint-html
· click to expand
Identification of HIV-Associated Gene Expression Biomarkers Using Machine Learning and Interpretable Artificial Intelligence | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Identification of HIV-Associated Gene Expression Biomarkers Using Machine Learning and Interpretable Artificial Intelligence View ORCID Profile Sultan Mahmud , View ORCID Profile Faruk Hossain doi: https://doi.org/10.1101/2025.05.08.652807 Sultan Mahmud 1 icddr ,b, 68 Shaheed Tajuddin Ahmed Sarani, Mohakhali, Dhaka 1212, Bangladesh Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sultan Mahmud For correspondence: smahmud{at}isrt.ac.bd Faruk Hossain 2 Institute of Statistical Research and Training, University of Dhaka , Dhaka, Bangladesh Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Faruk Hossain Abstract Full Text Info/History Metrics Preview PDF Abstract Despite advances in antiretroviral therapy (ART), the early and accurate diagnosis of Human Immunodeficiency Virus (HIV) infection remains a significant public health challenge. Traditional biomarkers, such as CD4+ T cell counts and viral load, are limited in capturing the complex biological mechanisms underlying HIV pathogenesis. This study proposes a machine learning (ML) and interpretable artificial intelligence (AI) framework to identify transcriptomic biomarkers for HIV diagnosis using immune cell–specific gene expression data. We utilized the GSE6740 dataset, which includes microarray profiles from CD4+ and CD8+ T cells of HIV-positive (treatment-naïve) and HIV-negative individuals. Feature selection was performed through differential gene expression analysis, weighted gene co-expression network analysis (WGCNA), and protein–protein interaction (PPI) network construction. Hub genes identified through these methods were used to train six supervised ML classifiers. The best-performing model was selected based on cross-validated metrics, including accuracy, Kappa, ROC, sensitivity, and specificity, and further interpreted using SHapley Additive exPlanations (SHAP) values. Seven co-expression modules were identified, with the red and green modules showing strong positive and negative correlations with HIV status, respectively. From the intersection of WGCNA modules, differentially expressed genes (DEGs), and PPI networks, ten hub genes were prioritized. Among the trained models, the regularized regression model (GLMNET) demonstrated the highest diagnostic performance (ROC = 0.97, accuracy = 91%). SHAP analysis highlighted GBP1, ISG15, OAS2, OAS1, and DDX60 as the most influential genes contributing to model predictions, thereby enhancing interpretability and biological relevance. By integrating transcriptomic profiling with interpretable ML, this study identifies novel gene-based biomarkers for HIV diagnosis and underscores the potential of explainable AI in advancing precision medicine approaches for infectious diseases. Introduction Human Immunodeficiency Virus (HIV) continues to pose a critical global public health challenge, with an estimated 39 million people living with the virus as of 2023 [ 1 ]. Despite significant advances in antiretroviral therapy (ART) that have markedly improved the survival and quality of life for people living with HIV (PLWH), substantial challenges remain in the areas of early diagnosis, personalized treatment, and long-term disease monitoring [ 2 ]. Current clinical practices primarily rely on biomarkers such as CD4+ T cell counts and plasma viral load to assess disease progression and treatment response. However, these conventional markers often fall short in capturing the complex and dynamic host–virus interactions, and they may not fully reflect the underlying biological processes involved in HIV pathogenesis [ 3 , 4 ]. In recent years, gene expression profiling has emerged as a promising tool for elucidating host responses to HIV infection. Studies have demonstrated that HIV triggers distinct transcriptional changes associated with immune activation, inflammation, and apoptosis [ 5 - 7 ]. However, many of these investigations have employed traditional statistical approaches that focus on a limited set of genes, which may hinder broader applicability and biological interpretability [ 8 ]. Furthermore, the integration of immune-specific transcriptomic data into robust diagnostic frameworks remains limited. The growing field of machine learning (ML) and interpretable artificial intelligence (AI) presents a promising opportunity to overcome these limitations. ML techniques excel at modeling complex, high-dimensional data and can detect subtle patterns in gene expression that are indicative of disease states [ 9 , 10 ]. Furthermore, the application of explainable AI methods such as SHapley Additive exPlanations (SHAP) allows researchers to interpret and quantify the impact of individual genes on model predictions, thereby enhancing both the transparency and biological interpretability of diagnostic models [ 11 ]. In this study, we propose an integrated ML and interpretable AI framework for identifying diagnostic biomarkers of HIV infection using immune-cell-specific transcriptomic data. We leveraged the publicly available GSE6740 dataset, which includes microarray gene expression profiles from CD4+ and CD8+ T cells of both HIV-positive (treatment-naïve) and HIV-negative individuals [ 12 ]. Our analytical pipeline incorporated differential gene expression analysis, weighted gene co-expression network analysis (WGCNA), and protein–protein interaction (PPI) network construction to identify and prioritize HIV-related genes. The most influential hub genes were then used to train a range of supervised ML models—logistic regression, support vector machines (SVM), k-nearest neighbors (kNN), decision trees, random forest (RF), and GLMNET. To ensure interpretability, SHAP values were computed to elucidate the contribution of each feature to model predictions. This study makes the following novel contributions: It integrates WGCNA, differential expression, and PPI network analysis for robust feature selection It applies multiple ML algorithms to diagnose HIV infection based on transcriptomic data. It incorporates SHAP analysis to ensure transparency and interpretability of predictive models. By combining computational modeling with biologically informed feature selection, this study aims to advance the development of accurate, interpretable, and clinically applicable diagnostic tools for HIV. Method Study Design and Data Sources This study employed a retrospective design to identify HIV-associated gene expression biomarkers by integrating machine learning and interpretable artificial intelligence (AI) techniques. Microarray gene expression data were obtained from the Gene Expression Omnibus (GEO) database. Specifically, the GSE6740 dataset [ 12 ], which includes both HIV-positive (case) and HIV-negative (control) samples along with relevant clinical metadata (e.g., age, treatment status, and duration of infection), was used for model development and training. Data Preprocessing The data were transposed to ensure that rows represented samples and columns represented genes, as required by the WGCNA framework. To reduce noise, a pre-filtering step was applied in which genes with zero variance were removed, and the top 11% of genes with the highest variance were retained. As shown in the gene variance curve (S1 Fig 1), the variance sharply drops and flattens beyond this threshold. A log2(TPM + 1) transformation was applied to normalize gene expression levels. Quality control was performed using the goodSamplesGenes function from the WGCNA package to retain only high-quality genes and samples. Phenotypic trait data-including gender, age, infection duration, and HIV status-were extracted and cleaned using regular expressions and string processing functions. Differential Gene Expression Analysis Differentially expressed genes (DEGs) between HIV-positive and HIV-negative individuals were identified using the limma package in R. Genes with an adjusted p-value 0.2 were considered significantly differentially expressed. Weighted Gene Co-expression Network Analysis (WGCNA) To identify gene modules associated with HIV status, WGCNA was performed using the 11% top most variable genes. Modules showing strong correlation with HIV (positive/negative) status were selected for further analysis. The intersection between the significant WGCNA modules and DEGs was used to identify key hub genes for model development. Gene Ontology (GO) Enrichment Analysis To gain insights into the biological significance of the selected genes, Gene Ontology (GO) enrichment analysis was performed using the enrichGO() function from the clusterProfiler R package. The analysis was conducted with the following parameters: the input genes were mapped using their ENTREZ IDs (keyType = “ENTREZID”) against the org.Hs.eg.db database. GO terms across all three ontologies—biological process (BP), molecular function (MF), and cellular component (CC)—were considered (ont = “ALL”). Multiple testing correction was applied using the Benjamini-Hochberg (BH) method (pAdjustMethod = “BH”), with a p-value cutoff of 0.05 and a q-value cutoff of 0.2. Gene IDs were converted into readable gene symbols to facilitate interpretation. KEGG Pathway Enrichment Analysis Kyoto Encyclopedia of Genes and Genomes (KEGG) pathway enrichment analysis was also performed to identify significantly enriched signaling pathways using the enrichKEGG() function. ENTREZ IDs were used for gene mapping against the hsa organism (Homo sapiens). the analysis was conducted with a p-value cutoff of 0.05. The results included measures such as GeneRatio, background ratio (BgRatio), fold enrichment, p-values, adjusted p-values, and q-values. PPI Network Analysis and Hub Gene Screening Protein–protein interaction (PPI) network analysis was performed using the STRING database (version 12.0) to construct a network based on important overlapping co-expressed genes [ 13 , 14 ]. The top ten hub genes within the PPI network were identified using the Maximum Clique Centrality (MCC) algorithm implemented in the CytoHubba plugin for Cytoscape (version 3.9.3) [ 15 ]. The MCC algorithm ranks genes based on their centrality within the network, enabling the identification of key hub genes. The final results, including the top-ranked hub genes, were visualized using Cytoscape. Development and assessment HIV diagnostic model The dataset contained the expression profiles of selected genes, clinical trait status (infected vs. uninfected), age, and other relevant variables, which were used to build machine learning models. A suite of supervised machine learning algorithms was applied to construct diagnostic models based on these features. The following models were trained using the caret package in R: Logistic Regression (Logistic) [ 16 ]: A generalized linear model using a binomial link function. Regularized Regression (GLMNET) [ 17 ]: Elastic net regularization via the glmnet method with preprocessing (centering and scaling). Support Vector Machine (SVM) with Radial Kernel [ 18 ]: Trained using the svmRadial method, with preprocessing. K-Nearest Neighbors (kNN) [ 19 ]: Classification based on feature proximity, with preprocessing. Decision Tree (DecisionTree) [ 20 ]: A recursive partitioning tree implemented via the rpart method. Random Forest (RF) [ 21 ]: A robust ensemble learning method utilizing multiple decision trees. Model performance was evaluated using 10-fold cross-validation repeated three times to ensure robust estimation and reduce variability. The evaluation metrics used to compare model performance included accuracy, Kappa statistic, area under the ROC curve (AUC), sensitivity, and specificity [ 20 ]. Once the best-performing model was identified based on these metrics, a hyperparameter tuning procedure was conducted to optimize the selected GLMNET model using the caret and glmnet packages in R. The alpha parameter determines the type of regularization (Ridge, Lasso, or Elastic Net), while the lambda parameter controls the strength of the regularization. A range of values for both hyperparameters was explored, with alpha values of 0 (Ridge), 0.5 (Elastic Net), and 1 (Lasso), and lambda ranging from 0.0001 to 1 on a logarithmic scale. A 10-fold cross-validation procedure was applied to ensure that the model’s performance was robust and generalized well to unseen data. The grid search process identified the best combination of alpha and lambda, optimizing the model based on accuracy.. Model Interpretation To enhance model interpretability, SHapley Additive exPlanations (SHAP) values were calculated for the top-performing model (GLMNET). SHAP provides global and local feature importance, offering insights into how each gene and clinical variable contributed to the classification outcome. Statistical Analysis and Visualization All analyses were performed using R (version 4.5) and Python (Jupyter-Notebook version 7.2.2). R packages used included caret, randomForest, e1071, glmnet, WGCNA, ggplot2 and SHAP. Summary statistics were presented in tables, while model performance and interpretability outputs were visualized using bar plots, ROC curves, and SHAP summary plots. Results Our study’s overall workflow is depicted in Fig. 1 . The dataset GSE6740 included expression profiles for 22283 genes, presenting the typical high-dimensionality challenge common in genomics research. To effectively manage this complexity and identify strong biomarker candidates for HIV infection, we applied several feature selection algorithms. Following feature selection, the development of AI-based diagnostic models was carried out using the most important (hub) genes. These hub genes were selected based on both their statistical significance and their potential biological relevance to disease progression. Download figure Open in new tab Fig 1. Workflow for identifying and validating diagnostic hub genes for HIV Feature Selection Weighted Gene Co-expression Network Modules WGCNA analysis was performed on the expression data of 2,451 genes, following the removal of genes with low expression variability (i.e., those outside the top 11% of variance) [ 22 ]. A soft-thresholding power of 9 was selected to achieve an optimal balance between scale-free topology fit and network connectivity, based on soft-thresholding power analysis-a critical step in WGCNA to construct a scale-free gene co-expression network (S1 Fig (2(a-b)). According to Fig 2 , seven modules (including the gray module) were identified and can be clustered into two group. A heatmap of module–trait relationship was generated to assess the association between each module and the clinical trait, HIV status (HIV+ and HIV−) ( Fig 3 ). This analysis revealed that the red and yellow modules had significant positive association with HIV status (red module: r = 0.54, p = 3e-04; yellow module: r = 0.50, p = 0.001), while the brown and green modules showed significant negative association (brown module: r = -0.4, p = 0.01; green module: r = -0.41, p = 0.009). Focusing on the gene modules, particularly the red strongest disease positive and blue module with the strongest disease negative correlation and most significant p-values, may provide deeper insights into the biological mechanisms of HIV infection and identify potential targets for diagnosis, prognostic assessment, and therapeutic development. Download figure Open in new tab Fig 2. Gene and module clustering dendrogram Download figure Open in new tab Fig 3. Module trait relationship Intersection of DEGs and Co-expression Modules A total of 198 differentially expressed genes (DEGs) were identified in the GSE6740 dataset (|logFC| > 0.2, adjusted p-value < 0.05), including 94 upregulated and 104 downregulated genes (right panel of Fig 4 ). As illustrated in the Fig, 170 co-expressed genes were identified in the red module and 274 in the green module. Of these, 66 genes from the red module and 27 from the green module overlapped with the DEGs identified in the GSE6740 dataset (left panel of Fig 4 ). These overlapping genes were designated as differentially co-expressed genes. Download figure Open in new tab Fig 4. Differentially expressed genes (DEGs) from GSE6740 dataset and intersection with yellow module genes Download figure Open in new tab Fig 5. Ten hub genes with ranks from PPI network analysis Gene Ontology (GO) and KEGG Pathway Analysis A total of 80 unique differentially co-expressed genes (from both the red and green modules) were used for Gene Ontology (GO) and KEGG pathway enrichment analysis. These 80 genes were selected from 93 by removing duplicate genes that were co-expressed in both modules. The enrichment results revealed highly significant GO terms, indicating strong activation of antiviral defense mechanisms in HIV-infected individuals (S1 Tables 1). Notably, the enrichment of negative regulators of viral processes suggests upregulation of host genes involved in controlling HIV replication. Key upregulated genes include ISG15, IFIT1, OAS1/2/3, RSAD2, MX1, TRIM22, PLSCR1, and BST2, all of which are interferon-inducible genes with established anti-HIV activity (S1 Fig 3). KEGG pathway enriched reflect the activation of shared antiviral immune responses-such as interferon-stimulated genes (ISGs), inflammatory cytokines, and cellular defense mechanisms (S1 Table 2). Although these pathways are labeled for different viruses (e.g., Influenza A, Hepatitis C), the shared components of host antiviral defense (e.g., IFITs, ISG15, MX1, OAS1-3) are likely upregulated in HIV infection as well. This suggests cross-pathogen immune activation—a signature of chronic HIV infection due to ongoing immune stimulation and inflammation. (S1 Fig 4). PPI Network Analysis and Hub Gene Identification PPI network analysis was performed by uploading selected 80 genes (both red and green) on the STRING website. Then the PPI network was imported in Cytoscape for hub gene screening (S1 Fig 6). Using MCC algorithm top ten genes with the highest scores were defined as Hub genes, including: DDX60, IFI44L, IFI44, IFIT1, IFIT3, RSAD2, OAS2, OAS1, ISG15, and GBP1 ( Fig 4 ). Identification of the Optimal HIV Diagnostic Model Machine learning models were developed using clinical information (such as age and treatment group) along with the expression profiles of ten hub genes to identify the most effective HIV diagnostic model. Among the six models evaluated-Logistic Regression, GLMNET, SVM, kNN, Decision Tree, and Random Forest - GLMNET emerged as the best-performing model overall. It achieved the highest mean ROC (0.99), indicating excellent discriminative ability, and also recorded the highest mean accuracy (0.92) across 30 resamples ( Table 1 ; Figs 7 – 8 ). In terms of specificity (0.92) and kappa (0.72), GLMNET outperformed all other models, reflecting both strong classification performance and substantial agreement beyond chance. While its sensitivity (0.733) was slightly lower than that of Logistic Regression (0.97), GLMNET maintained a better balance between sensitivity and specificity, making it a more reliable model overall. Although Random Forest and Decision Tree also performed well, GLMNET provided the most consistent and robust performance across all key evaluation metrics, making it the most suitable model for the dataset. The best-performing GLMNET model was obtained with the hyperparameters alpha = 0 and lambda = 0.29, identified through 10-fold cross-validation using a grid search over a range of alpha and lambda values. This optimal combination was selected based on the highest cross-validated accuracy from the training results. View this table: View inline View popup Download powerpoint Table 1. Average accuracy, Kappa, ROC (AUC), sensitivity, and specificity for each model estimated using 10-fold cross-validation Download figure Open in new tab Fig 6. Average accuracy, Kappa, ROC (AUC), sensitivity (Sens), and specificity (Spac) for each model, along with 95% confidence intervals, estimated using 10-fold cross-validation. Download figure Open in new tab Fig 7. ROC curve comparison among selected machine learning models Interpretation and Visualization of the HIV Diagnostic Model This SHAP analysis provided insights into the key drivers of the model’s predictions. The model’s average predicted probability was 0.75, while the analyzed sample showed a higher prediction of 0.90 (S1 Fig 6(a-f)), indicating a positive deviation primarily influenced by specific features. The most impactful features were GBP1, ISG15, OAS2, OAS1, and DDX60, which consistently pushed predictions higher. GBP1 emerged as the strongest contributor overall. Age also influenced predictions both positively and negatively, while the treatment group had lowest impact ( Fig 8 (a-b) , S1 Fig 6 (a-f)). Download figure Open in new tab Fig 8. (a) Summary of SHAP Values Across Features. (b) Global feature importance based on absolute shape value Discussion This study presents a novel integrative framework that combines immune cell–specific transcriptomic data with interpretable machine learning to identify robust and biologically relevant biomarkers for HIV infection. Traditional clinical biomarkers, such as CD4+ T cell count and viral load, while essential, do not fully capture the molecular complexity of HIV pathogenesis [ 3 ]. Our findings emphasize that transcriptomic signatures derived from specific immune cells, analyzed through systems biology and machine learning pipelines, can provide deeper insights into host– virus interactions [ 23 , 24 ]. The WGCNA identified co-expression modules significantly associated with HIV status, particularly the red and green modules, which showed the strongest correlation. Integration with differentially expressed genes (DEGs) and protein–protein interaction (PPI) networks allowed us to isolate hub genes that are not only statistically significant but also biologically central to the underlying network, potentially reflecting key drivers of HIV-related immune dysregulation. These genes enriched Gene Ontology (GO) terms related to immune response, viral processes, and apoptosis, aligning with known mechanisms of HIV pathology [ 25 ]. Among the various machine learning (ML) classifiers developed, the GLMNET model-optimized through hyperparameter tuning-achieved the best diagnostic performance, with a high area under the curve (AUC). This is consistent with prior findings that regularized models like GLMNET are well-suited for high-dimensional omics data [ 26 , 27 ]. Importantly, SHapley Additive exPlanations (SHAP) values enhanced model interpretability by identifying GBP1, ISG15, OAS2, OAS1, and DDX60 as the most impactful genes for differentiating HIV-positive from HIV-negative individuals. Notably, GBP1 has also been reported as a novel biomarker for chronic inflammatory diseases [ 28 ], while previous studies suggest that ISG15 and OAS2 may play important roles in the anti-HIV-1 response of CD4+ and CD8+ T cells [ 29 ]. Our study underscores the utility of combining biologically informed feature selection methods with interpretable AI to enhance the diagnostic landscape for HIV. By using only peripheral immune cell gene expression profiles, we offer a minimally invasive yet highly informative approach to support early HIV diagnosis. Additionally, the identified hub genes may serve as potential targets for future therapeutic research or as candidates for longitudinal monitoring of treatment response. However, the study has limitations. First, the findings are based on a single dataset (GSE6740) with a relatively small sample size, which may limit generalizability. Future studies should validate the model on larger and more diverse cohorts. Second, while SHAP provides model interpretability, experimental validation of the selected biomarkers is necessary to establish causal relevance. Lastly, the transcriptomic data represents a static snapshot; integrating longitudinal data or single-cell sequencing could further enrich the diagnostic model and improve precision. In conclusion, our study demonstrates that integrating transcriptomic profiling with machine learning and explainable AI can lead to the discovery of clinically meaningful and interpretable biomarkers for HIV diagnosis. This approach has the potential to complement existing diagnostic tools and contribute to more personalized and timely HIV care strategies. Data availability statement Data file is publicly available from the Gene Expression Omnibus (GEO) database ( https://www.ncbi.nlm.nih.gov/gds ). All code files are available from Open Science Framework repository at https://doi.org/10.17605/OSF.IO/WMQ7K . S1: supportive information References 1. ↵ UNAIDS . Global HIV & AIDS statistics—Fact sheet 2023 . Available from: https://www.unaids.org/en/resources/fact-sheet#:∼:text=Global%20HIV%20statistics,AIDS%2Drelated%20illnesses%20in%202023. 2. ↵ World Health Organization . Consolidated guidelines on HIV prevention, testing, treatment, service delivery and monitoring 2021 . Available from: https://www.who.int/publications/i/item/9789240031593 . 3. ↵ Deeks SG , Lewin SR , Havlir DV . The end of AIDS: HIV infection as a chronic disease . The lancet . 2013 ; 382 ( 9903 ): 1525 – 33 . OpenUrl 4. ↵ Mellors JW , Munoz A , Giorgi JV , Margolick JB , Tassoni CJ , Gupta P , et al. Plasma viral load and CD4+ lymphocytes as prognostic markers of HIV-1 infection . Annals of internal medicine . 1997 ; 126 ( 12 ): 946 – 54 . OpenUrl CrossRef PubMed Web of Science 5. ↵ Rotger M , Dalmau J , Rauch A , McLaren P , Bosinger SE , Martinez R , et al. Comparative transcriptomics of extreme phenotypes of human HIV-1 infection and SIV infection in sooty mangabey and rhesus macaque . The Journal of clinical investigation . 2011 ; 121 ( 6 ): 2391 – 400 . OpenUrl CrossRef PubMed Web of Science 6. Bosinger SE , Li Q , Gordon SN , Klatt NR , Duan L , Xu L , et al. Global genomic analysis reveals rapid control of a robust innate response in SIV-infected sooty mangabeys . The Journal of clinical investigation . 2009 ; 119 ( 12 ): 3556 – 72 . OpenUrl CrossRef PubMed Web of Science 7. ↵ van’t Wout AlB , Lehrman GK , Mikheeva SA , O’Keeffe GC , Katze MG , Bumgarner RE , et al. Cellular gene expression upon human immunodeficiency virus type 1 infection of CD4+-T-cell lines . Journal of virology . 2003 ; 77 ( 2 ): 1392 – 402 . OpenUrl Abstract / FREE Full Text 8. ↵ Salgado M , Rallón NI , Rodés B , López M , Soriano V , Benito JM . Long-term non-progressors display a greater number of Th17 cells than HIV-infected typical progressors . Clinical immunology . 2011 ; 139 ( 2 ): 110 – 4 . OpenUrl CrossRef PubMed 9. ↵ Libbrecht MW , Noble WS . Machine learning applications in genetics and genomics . Nature Reviews Genetics . 2015 ; 16 ( 6 ): 321 – 32 . OpenUrl CrossRef PubMed 10. ↵ Kourou K , Exarchos TP , Exarchos KP , Karamouzis MV , Fotiadis DI . Machine learning applications in cancer prognosis and prediction . Computational and structural biotechnology journal . 2015 ; 13 : 8 – 17 . OpenUrl CrossRef 11. ↵ Lundberg SM , Lee S-I. A unified approach to interpreting model predictions . Advances in neural information processing systems . 2017 ; 30 . 12. ↵ Hyrcza MD , Kovacs C , Loutfy M , Halpenny R , Heisler L , Yang S , et al. Distinct transcriptional profiles in ex vivo CD4+ and CD8+ T cells are established early in human immunodeficiency virus type 1 infection and are characterized by a chronic interferon response as well as extensive transcriptional changes in CD8+ T cells . J Virol . 2007 ; 81 ( 7 ): 3477 – 86 . Epub 20070124. doi: 10.1128/jvi.01552-06 . PubMed PMID: 17251300 ; PubMed Central PMCID: PMCPMC1866039 . OpenUrl Abstract / FREE Full Text 13. ↵ Szklarczyk D , Kirsch R , Koutrouli M , Nastou K , Mehryary F , Hachilif R , et al. The STRING database in 2023: protein–protein association networks and functional enrichment analyses for any sequenced genome of interest . Nucleic acids research . 2023 ; 51 ( D1 ): D638 - D46 . OpenUrl CrossRef PubMed 14. ↵ Data were retrieved from STRING (v12.0) 2025 . Available from: https://version-12-0.string-db.org/cgi/input?sessionId=bHuaBbafe7bC&input_page_show_search=on . 15. ↵ Shannon P , Markiel A , Ozier O , Baliga NS , Wang JT , Ramage D , et al. Cytoscape: a software environment for integrated models of biomolecular interaction networks . Genome research . 2003 ; 13 ( 11 ): 2498 – 504 . OpenUrl Abstract / FREE Full Text 16. ↵ Nick TG , Campbell KM . Logistic regression . Topics in biostatistics . 2007 : 273 – 301 . 17. ↵ Ogutu JO , Schulz-Streeck T , Piepho H-P , editors. Genomic selection using regularized linear regression models: ridge regression, lasso, elastic net and their extensions . BMC proceedings ; 2012 : Springer . 18. ↵ Razaque A , Ben Haj Frej M , Almi’ani M , Alotaibi M , Alotaibi B. Improved support vector machine enabled radial basis function and linear variants for remote sensing image classification . Sensors . 2021 ; 21 ( 13 ): 4431 . OpenUrl CrossRef PubMed 19. ↵ Mucherino A , Papajorgji PJ , Pardalos PM , Mucherino A , Papajorgji PJ , Pardalos PM . Knearest neighbor classification . Data mining in agriculture . 2009 : 83 – 106 . 20. ↵ Mahmud S , Mohsin M , Muyeed A , Nazneen S , Sayed MA , Murshed N , et al. Machine learning approaches for predicting suicidal behaviors among university students in bangladesh during the covid-19 pandemic: A cross-sectional study . Medicine . 2023 ; 102 ( 28 ): e34285 . OpenUrl CrossRef PubMed 21. ↵ Qi Y. Random forest for bioinformatics . Ensemble machine learning: Methods and applications . 2012 : 307 – 23 . 22. ↵ Wang M , Wang L , Pu L , Li K , Feng T , Zheng P , et al. LncRNAs related key pathways and genes in ischemic stroke by weighted gene co-expression network analysis (WGCNA) . Genomics . 2020 ; 112 ( 3 ): 2302 – 8 . OpenUrl CrossRef PubMed 23. ↵ Ghosh D , Chakraborty S , Kodamana H , Chakraborty S. Application of machine learning in understanding plant virus pathogenesis: trends and perspectives on emergence, diagnosis, hostvirus interplay and management . Virology Journal . 2022 ; 19 ( 1 ): 42 . OpenUrl CrossRef PubMed 24. ↵ Taroni JN , Grayson PC , Hu Q , Eddy S , Kretzler M , Merkel PA , et al. MultiPLIER: A Transfer Learning Framework for Transcriptomics Reveals Systemic Features of Rare Disease . Cell Systems . 2019 ; 8 ( 5 ): 380 - 94.e4 . doi: 10.1016/j.cels.2019.04.003 . OpenUrl CrossRef PubMed 25. ↵ Ivanov S , Lagunin A , Filimonov D , Tarasova O. Network-based analysis of OMICs data to understand the HIV–host interaction . Frontiers in Microbiology . 2020 ; 11 : 1314 . OpenUrl CrossRef PubMed 26. ↵ Yu W , Park T. AucPR: an AUC-based approach using penalized regression for disease prediction with high-dimensional omics data . BMC genomics . 2014 ; 15 : 1 – 12 . OpenUrl CrossRef PubMed 27. ↵ Determan Jr CE . Optimal algorithm for metabolomics classification and feature selection varies by dataset . International journal of biology . 2015 ; 7 ( 1 ): 100 . OpenUrl 28. ↵ Hammon M , Herrmann M , Bleiziffer O , Pryymachuk G , Andreoli L , Munoz LE , et al. Role of guanylate binding protein-1 in vascular defects associated with chronic inflammatory diseases . Journal of cellular and molecular medicine . 2011 ; 15 ( 7 ): 1582 – 92 . OpenUrl CrossRef PubMed 29. ↵ Gao L , Wang Y , Li Y , Dong Y , Yang A , Zhang J , et al. Genome-wide expression profiling analysis to identify key genes in the anti-HIV mechanism of CD4+ and CD8+ T cells . Journal of medical virology . 2018 ; 90 ( 7 ): 1199 – 209 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted May 12, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Identification of HIV-Associated Gene Expression Biomarkers Using Machine Learning and Interpretable Artificial Intelligence Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Identification of HIV-Associated Gene Expression Biomarkers Using Machine Learning and Interpretable Artificial Intelligence Sultan Mahmud , Faruk Hossain bioRxiv 2025.05.08.652807; doi: https://doi.org/10.1101/2025.05.08.652807 Share This Article: Copy Citation Tools Identification of HIV-Associated Gene Expression Biomarkers Using Machine Learning and Interpretable Artificial Intelligence Sultan Mahmud , Faruk Hossain bioRxiv 2025.05.08.652807; doi: https://doi.org/10.1101/2025.05.08.652807 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7619) Biochemistry (17639) Bioengineering (13865) Bioinformatics (41854) Biophysics (21405) Cancer Biology (18544) Cell Biology (25432) Clinical Trials (138) Developmental Biology (13356) Ecology (19863) Epidemiology (2067) Evolutionary Biology (24287) Genetics (15587) Genomics (22465) Immunology (17701) Microbiology (40300) Molecular Biology (17142) Neuroscience (88440) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4814) Physiology (7633) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4285) Systems Biology (9809) Zoology (2268)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.