ImmuneFM: Pre-training Foundation Model from Cytometry Data for Immunology Research

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Immunology is an essential field in the biomedicine domain, which plays an important role in oncology, vaccines, infection, etc. With the increasing amount of data available in immunology and artificial intelligence technique development, there is a need to develop data-driven AI methods in the field. However, the data from various immunology studies is very hard to integrate into an AI-ready dataset due to the lack of a standard. Moreover, independent immunology studies’ data lacks enough labels to train the supervised model. Motivated by these challenges, we curated a large-scale AI-ready cytometry dataset for immunology from the publicly available ImmPort portal. We design the framework to pre-train a foundation model, ImmuneFM, on the cytometry dataset. ImmuneFM can be applied to a wide range of downstream immunology diseases with fine-tuning on a limited number of labeled samples. The experiment results on eight downstream tasks demonstrate the superior performance of ImmuneFM compared to baseline deep learning and traditional methods.
Full text 55,208 characters · extracted from preprint-html · click to expand
ImmuneFM: Pre-training Foundation Model from Cytometry Data for Immunology Research | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results ImmuneFM: Pre-training Foundation Model from Cytometry Data for Immunology Research Sirui Ding , Sanchita Bhattacharya , Atul J. Butte doi: https://doi.org/10.1101/2025.07.09.664020 Sirui Ding 1 Bakar Computational Health Sciences Institute, University of California , San Francisco, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: sanchita.bhattacharya{at}ucsf.edu sirui.ding{at}ucsf.edu Sanchita Bhattacharya 1 Bakar Computational Health Sciences Institute, University of California , San Francisco, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: sanchita.bhattacharya{at}ucsf.edu sirui.ding{at}ucsf.edu Atul J. Butte 1 Bakar Computational Health Sciences Institute, University of California , San Francisco, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Preview PDF Abstract Immunology is an essential field in the biomedicine domain, which plays an important role in oncology, vaccines, infection, etc. With the increasing amount of data available in immunology and artificial intelligence technique development, there is a need to develop data-driven AI methods in the field. However, the data from various immunology studies is very hard to integrate into an AI-ready dataset due to the lack of a standard. Moreover, independent immunology studies’ data lacks enough labels to train the supervised model. Motivated by these challenges, we curated a large-scale AI-ready cytometry dataset for immunology from the publicly available ImmPort portal. We design the framework to pre-train a foundation model, ImmuneFM, on the cytometry dataset. ImmuneFM can be applied to a wide range of downstream immunology diseases with fine-tuning on a limited number of labeled samples. The experiment results on eight downstream tasks demonstrate the superior performance of ImmuneFM compared to baseline deep learning and traditional methods. Introduction Immunology is a foundational discipline in the biomedicine domain 1 , playing a critical role in advancing our understanding and treatment of diverse diseases - from infectious diseases 2 to autoimmune disorders 3 and oncology 4 . In recent years, transformative breakthroughs have emerged in this field. For example, mRNA COVID-19 vaccines demonstrate quick immune-based strategies in responding to the global pandemic 5 . Immunology also revolutionizes cancer treatment, e.g., the CAR-T cell therapy, which engineers the patient’s T cell to target cancer cells 6 . The immune system’s complexity, with its intricate network of cells, signaling pathways, and molecular interactions, necessitates advanced tools for comprehensive analysis. Immunology research generates vast volumes of data from observational and clinical trial studies. The ImmPort data portal 7 , a publicly accessible repository for immunology-related studies, currently hosts 1,256 studies spanning 176 diseases and includes over 7.5 million experimental results (as of April 2025). This explosion of immunological data presents a timely opportunity to leverage artificial intelligence (AI) for both clinical diagnostics and scientific discovery. Recent advancements in AI have shown great promise in decoding the complexities in immunology, offering new insights into immune responses and potential therapeutic targets 8 . For instance, AI can be used to discover new biomarkers for immune oncology 9 ; AI can also be applied in drug discovery and drug repurposing 10 , 11 . However, the application of AI in immunology still faces significant challenges, particularly when dealing with independent studies involving very few subjects. One of the primary challenges in applying AI to immunology is the scarcity of large, well-curated datasets and the lack of standardization of immunology data. Unlike other fields where massive, labeled datasets are readily available, immunological studies often involve small patient cohorts due to the rarity of certain conditions, ethical considerations, and the high cost of data collection and labeling 12 . This data scarcity limits the ability of traditional AI models to learn robust patterns and make accurate predictions 13 . The other challenge is the heterogeneity of immune responses, which makes it hard for a supervised training AI model to be adapted to various types of diseases 14 . So, these two challenges make it a dilemma that immunologists need to train an AI model for each study, but each study lacks enough labeled data. These challenges highlight the need for a new paradigm in how AI is applied to immunology 15 , moving from models trained on isolated, small datasets to those that can integrate and learn from diverse sources of immunological data. With the rapidly increasing data accessibility in the public domain and promising performance of self-supervised learning, the foundation model becomes the new paradigm in the biomedical domain 16 . A foundation model is a general AI model pre- trained on large-scale and wide-range datasets, which can support various downstream tasks. Large language model (LLM) is one type of foundation model 17 . For example, a biomedicine foundation model can be used for disease diagnosis, clinical decision support, scientific discovery, etc 18 . Driven by big data and large-scale models, foundation models have shown strong power in the biomedicine domain. For example, Steinberg et al. designed MOTOR, a foundation model pre-trained on structured EHR data 19 , Wang et al. pre-trained a pathology foundation model which can be applied for cancer studies 20 , Yang et al. pre-trained GatorTron, a large language model on clinical text 21 . In the field of immunology, Kim et al. 22 pre-trained a masked autoencoder on the cytometry data specifically for COVID-19 analysis. However, there remains a lack of a generalizable immunology foundation model capable of addressing a broader spectrum of diseases and conditions, such as oncology, transplantation, and autoimmune disorders. In this work, we introduce ImmuneFM , a foundation model pre-trained on large-scale cytometry data from 53 immunology studies in the ImmPort repository, spanning eight domains including vaccine response, autoimmune disease, oncology, transplantation, infection, allergy, preterm birth, and general immune response. With cytometry accounting for nearly 70% of ImmPort results, we curated a dataset of over 100 million single cells and 203 commonly used panel markers. ImmuneFM follows a two-stage framework: during pre-training, raw cytometry data are normalized and masked for self- supervised learning using a Transformer model; during fine-tuning, cell embeddings are aggregated and passed to a multi-layer perceptron (MLP) for classification. We evaluated ImmuneFM on eight downstream tasks, such as liver cancer staging 23 , allergy prediction 24 , autoimmune disease classification 25 , and HIV/COVID diagnosis 26 – 28 , where it consistently outperformed baseline models. Further interpretability analysis at the cell and marker levels enabled both validation of known findings and discovery of new insights across three case studies, demonstrating ImmuneFM’s potential for both clinical application and scientific discovery. In summary, we have three-fold contributions in this study: We curated a large-scale AI-ready cytometry dataset covering a wide range of focusing immunology areas, from oncology to transplant. We propose and pre-train ImmuneFM, a foundation model for the immunology field to support clinical diagnosis and scientific discovery. We evaluate the ImmuneFM on 8 downstream tasks of different diseases, which demonstrate the superior accuracy and interpretability of the model. Methods Data preparation Pre-training Dataset curation We built the cytometry dataset from the ImmPort platform, which is an open-source database for immunology studies. As shown in Figure 2 , we integrate the cytometry data collected from 53 different studies covering 8 focusing areas on immunology, including vaccine, COVID-19, oncology, etc. Each immunology study used different panel markers in the cytometry assay, so we manually created a commonly used panel marker set, which contains 203 frequently used markers, e.g., CD4, CD8, etc. Due to the extremely high volume of cytometry data in the 53 studies, we sample 10% of the single cells from each subject in the study. The final curated cytometry dataset is shaped in a matrix, where each row represents a single cell, and each column represents a panel marker. The detailed marker name and study ID of the pre-training dataset are presented in Figure. 2 . Download figure Open in new tab Figure 1. Workflow overview of ImmuneFM. 53 public studies selected from the ImmPort portal were used to integrate a large-scale AI-ready cytometry dataset. ImmuneFM was pre-trained on this dataset and fine-tuned for downstream sample classification tasks. Download figure Open in new tab Figure 2. Study and marker details of the curated cytometry dataset for pre-training. There are 53 studies from ImmPort that are included, covering 203 unique panel markers. Pre-processing Because the studies are conducted by different labs on different samples, we apply the widely used normalization function to transform the single-cell data in cytometry 29 . Because we need to embed each marker, which requires the marker value to be an integer, we linearly scale the expression value with MinMaxScaler and then bin the values into discrete integers inspired by NLP techniques 30 and gene expression binning 31 . The ImmuneFM model Transformer backbone We built the immunology foundation model, the ImmuneFM, with the Transformer backbone. The input data is a cytometry matrix that can be embedded like the word token embedding in the BERT model. Meanwhile, the order of marker columns doesn’t influence the semantic meaning of the cytometry matrix. Thus, we don’t need a positional embedding like an NLP task. Because the cytometry marker number is not like the gene number, which can be nearly 20,000 31 , the commonly used Transformer 32 can achieve competitive performance with a 203-panel marker length. The core formulas for multi-head self-attention are as follows: where Q,J,V are the query, key, and value vectors. The ,,wi is a single attention Panel marker embedding. Like the word embedding to process text 33 , we are also inspired to embed each marker with a learnable representation. Because the embedding can ensure that similar markers have close and similar representations. We employ a trainable embedding layer to encode each panel marker with a look-up table. We can learn the inter-marker relations with this embedding, and we name this layer “Marker2Vec” as follows. where each panel marker m is mapped to a vector cm via mw’k,’2,;,c. Marker expression embedding. The value of each panel marker reflects the expression levels of specific proteins or molecules. The marker expression can be considered as the existing occurrence of different biomarkers from a single cell. Inspired by this, we follow the conventional method to convert the continuous value of a marker expression into a discrete integer value by binning it into an N -dimensional vector that can be input into ImmuneFM as follows. where xm is the expression value of marker m. Sample/bulk level representation. In the cytometry analysis, the sample or bulk classification is the focusing task, which is usually used for disease diagnosis, treatment effectiveness analysis, etc. A sample contains a bunch of single cells, e.g., 10k cells for each sample. We will encode each cell with the pre-trained backbone and aggregate the embedding into a sample-level tensor with mean pooling as follows. where N is the number of cells in a sample. Sample classification head. Then, a task-specific head, e.g., several linear layers, will be applied after the Transformer backbone to process the sample-level tensor and output the classification results. The classification head can be represented as follows. where the y’ is the model output from a multi-layer perceptron ("LPo. Pre-train and fine-tune Self-supervised learning with masking. In this work, we use the masking- regeneration task to pre-train the ImmuneFM. We will randomly mask the non-zero items in the input cytometry matrix and reconstruct the masked items with the remaining values in the matrix from the Transformer backbone output. Cross-entropy is used as the loss function to optimize as follows: where N is each sample’s cell number, " is the number of markers, which is 203 in the and 1th cell) in the original cytometry matrix and reconstructed output from the Supervised fine-tuning on downstream tasks. After pre-training the backbone Transformer, we will add a task-specific head to process the outputs. This task-specific head can be several layers of an MLP. In this work, we apply the MLP to output the classification results. We used cross-entropy to optimize the ImmuneFM as follows. where yi,y∼i indicate the ground truth and output from the model. Model interpretation We use the gradient-based class activation mapping (Grad-CAM) to calculate the importance of each cell. To interpret the model’s prediction, we compute the gradient of the predicted class score with respect to each cell embedding. These gradients are then combined with the original embeddings to produce importance scores for individual cells. This highlights which cells contributed most to the model’s decision, providing biological insights at the single-cell level. The interpretation calculation process can be represented as follows. where vi computes the gradient of class y∼i w.r.t. each cell embedding csample . The importance score i for each cell is calculated by multiplying the gradient the the embedding. Baseline methods FlowSOM + GBDT. We apply FlowSOM to cluster the cells in a sample. And then calculate statistical features from these clusters, including cluster percentage in the sample, mean fluorescence intensity (MFI) of panel markers. Then we used these derived features to train and test a GBDT classifier. 34 CNN. We compare with the CNN model specifically designed for cytometry sample classification, which is proposed by Hu et al. 29 . We follow the hyperparameter settings as described in the manuscript. Transformer. We compare with the Transformer backbone model without pre-training on the large-scale cytometry dataset. The hyperparameter settings for the plain Transformer model are the same as ImmuneFM. Implementation details For the pre-training of ImmuneFM , the training epoch is set to 100, and the learning rate is 1e-4. The batch size for pre-training is 1024 cells. The train-validation split is 0.95/0.05. The Transformer’s dimensions, depth, and heads are 128, 6, and 8, respectively. The bin number is 10. The masking ratio is 0.25. We use the Adam optimizer to pre-train the ImmuneFM. For the fine-tuning of ImmuneFM, the epoch is set to 50, and the learning rate is 1e-4 with the Adam optimizer. The cell number in each sample is set to 512, 256, and 128. The batch size for pre-training is 4. For the baseline models, the DeepCNN model follows the implementation and hyperparameter settings of Hu et al.’s work 29 . The Transformer model is the same architecture as ImmuneFM. FlowSOM is implemented with the Python package by Couckuyt et al 34 . The GBDT model is implemented by the Scikit Learn Python package. All the models are implemented with Python and packages including PyTorch, Scikit Learn, etc. The training and testing of the model are conducted on the UCSF Wynton platform. Downstream task curation We evaluated the ImmuneFM across eight downstream tasks as shown in Table 1 spanning various human immunology diseases, including a cross-species validation using a mouse dataset. These tasks were derived from the publicly available datasets in ImmPort, including SDY1733 (liver cancer), SDY2015(peanut allergy), SDY997(lupus nephritis), SDY1708(COVID-19), SDY2011 (COVID-19), SDY1535 (HIV), SDY788 View this table: View inline View popup Table 1. Statistical characteristics of downstream task studies. (kidney transplant), and SDY1108 (cancer). Evaluation metrics For the binary classification task, we use the area under the receiver operating characteristic curve (AUROC) as the evaluation metric. For the multi-class classification task, we use balanced accuracy (Bacc) as the evaluation metric. The formulas to calculate AUROC and Bacc are as follows. Benchmarking against state-of-the-art methods We benchmark ImmuneFM on eight downstream tasks against state-of-the-art baseline methods and under different hyperparameter settings (See Methods). We have several observations and insights from the results in Figure 3 (a), (b), and (c). Download figure Open in new tab Download figure Open in new tab Figure 3. Sample classification performance on 8 different immunology diseases and conditions. For binary classification, AUROC is the metric for evaluation. For multi-class classification, balanced accuracy (Bacc) is the metric for evaluation. As shown in Figure 3 (a), we found that ImmuneFM significantly outperforms DeepCNN 29 , Transformer 31 , and FlowSOM+GBDT 34 across the downstream sample classification tasks. Firstly, we observed that the CNN model has the lowest performance, which indicates the drawback of the CNN model under the limited number of labeled samples. Secondly, the Transformer without pre-training outperforms the CNN model, which implies a more suitable model architecture in cytometry classifications. Compared to the pre-trained ImmuneFM, the plain Transformer model has significantly worse performance, indicating the necessity and boosting effect of pre- training. Thirdly, the traditional method, such as FlowSOM+GBDT, achieved surprisingly competitive performance. The traditional method with feature engineering, like FlowSOM, outperforms the deep learning models, including CNN and Transformer. It makes sense especially under the circumstances of limited labeled samples. The traditional method with feature engineering needs fewer labeled samples than supervised deep learning. As shown in Figure 3 (b), the insight is that the cell number in each sample/bulk didn’t have a significant effect on the performance of deep learning models. We can observe that the ImmuneFM with much fewer cells, e.g., 256 or 128 cells in each bulk, also has competitive performance compared to 512 cells as input. This phenomenon is also observed in the CNN model. However, the cell number in each bulk has a larger effect on the FlowSOM+GBDT. We can observe that the performance of FlowSOM+GBDT drops with fewer cells in each sample. This is mainly due to the traditional method relies heavily on feature engineering which is sensitive to the number of cells. As shown in Figure 3 (c), we can find that ImmuneFM is robust with different pre-processing hyperparameters, e.g., the arcsinh factor. This result demonstrates the robustness of the sensitivity of ImmuneFM to the hyperparameters. Interpretable analysis for scientific results To further understand how ImmuneFM makes predictions and to explore its capacity for scientific insight, we conducted an interpretable analysis at both the marker and single- cell resolution. Building on the strong classification performance, this analysis aimed to explain model decisions and uncover biologically meaningful patterns. We focused on three case studies—COVID-19, HIV, and liver cancer—and used class activation mapping to visualize the importance of individual cells and markers in the cytometry input matrix ( Figure 4 ). At the marker level, we computed the average contribution of each marker across positive and negative cohorts and compared expression levels to highlight differentially informative biomarkers ( Figure 5 ). At the cell level, we leveraged these key markers to identify immune cell populations and assess their relevance to sample classification ( Figure 6 ). This dual-level interpretation not only aligned with findings from the literature but also revealed novel cellular and molecular features, underscoring ImmuneFM’s potential for enabling data-driven discovery in immunology. We provided clinical and scientific insights based on the marker and cell interpretation results. These insights are either validated with the previous literature with scientific evidence or new discoveries that haven’t been widely reported in previous works. Download figure Open in new tab Download figure Open in new tab Figure 4. Heatmap on input cytometry matrix. Each row represents a single cell, and each column represents a unique panel marker. Color indicates the importance of each cell or marker. Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab Figure 5. Importance and expression of panel markers in positive and negative patient groups. The subfigure (1) indicates the importance of the marker that contributes to the classification results, and subfigure (2) shows the expression level of important markers in different subgroups. Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab Figure 6. Bivariate plot of cell clusters. Each dot represents a single cell. The target cell types are identified by a red or blue circle. Color of the dot indicates importance. Case study on COVID-19 (whole blood) SDY2011 data (COVID-19, human). This dataset is derived from a COVID-19 study 27 . The task is a binary classification of COVID-19 positive and negative. The samples are collected from serum, whole blood, and PBMC. There are 185 labeled samples. The data is available at SDY2011 on ImmPort. Important marker interpretation. As shown in Figure 5 (a), the top 20 important markers are presented with importance calculated from the model and their expression in COVID-positive and COVID-negative groups. It reveals key patterns in the human immune response to SARS-CoV-2 infection. The prominence of myeloid markers (CD11b, CD11c, CD14) addresses the importance of monocyte activation in COVID-19 pathogenesis 39 . Notably, CD16 and CD66B show key patterns that potentially indicate neutrophil activity 40 . The prominence of CD39 suggests profound immunoregulatory alterations, possibly through adenosine-mediated suppression of effector functions 41 . CD45 maintains consistent importance as a pan-leukocyte activation marker that is important in both negative and positive groups. The high importance of CD64 points to Fcγ receptor-mediated mechanisms potentially leading to antibody-dependent enhancement. The pattern of CD123 possibly reflects plasmacytoid dendritic cell dysfunction, which has been validated in the previous study 42 . From the expression difference as shown in Figure 5(a) , notably, the CD39 and CD64 expression in the negative groups is significantly lower. CD39’s significantly lower expression in negative groups highlights its disease-specific role in COVID-19 41 . This clear expression pattern makes it a strong candidate for diagnosis and prognosis monitoring, especially for identifying immune exhaustion in severe cases. CD64’s minimal expression in negative samples also establishes it as a highly specific COVID-19 infection biomarker. Its significant upregulation indicates the profound monocyte/macrophage activation 43 . Important cell type interpretation. Based on the important markers identified by ImmuneFM, we identified the important cell types based on these markers. Here we present and analyze a representative and important cell type, neutrophil cells, as shown in Figure 6(a) . Our study utilized a combination of neutrophil-specific markers (CD16, CD66B, CD11B, and CD14) to identify the neutrophil and analyze its role in COVID-19. The results demonstrate a significant increase in neutrophil levels in positive groups, from a cell proportion of 27% to 46%. Neutrophils are among the first responders to viral infections 44 , consistent with their established role in early immune defense mechanisms. The observed neutrophil elevation suggests their significant involvement in the host response to SARS-CoV-2 infection, which has been validated 27 . This quantitative change in neutrophil prevalence underscores their importance in COVID-19 pathogenesis and worth further investigation into their specific functional states during infection. Clinical and scientific insights. Based on the analysis at the marker and cell levels, we summarized the key insights from clinical and scientific perspectives as follows. CD39 expression is significantly higher in COVID-19 patients, which could be a promising biomarker. This point has been validated by several previous scientific studies. In the work of GB da Silva et al. 45 , they observed the increased expression of CD39 in moderate and severe cases, which has been stated as one of the key messages from their work. In the work of E Diaz-Garcia et al. 41 , they aim to quantify the relation between CD39 expression and severity in COVID-19 patients. They found that CD39 could be a promising biomarker for COVID-19 severity, and the overexpression of CD39 might be the reason for the disorder of thromboinflammation. CD64 could be an important COVID-19 biomarker with higher expression in positive patients. This insight has also been validated by previous literature. In the study by M. Karawajczyk et al. 43 , CD64 is an early marker for COVID-19 infection and an indicator of severe cases. In the work of P. Bourgoin et al. 46 , they found that CD64 can be used to classify bacterial infection, COVID-19 infection, or other viral infections. This finding has the potential clinical application in the emergency department. Neutrophil cell increases in COVID-19 patients and play an important role in the immune response. This point is also validated in the corresponding literature of SDY2011 27 . In the work by E McKenna et al. 47 , severe COVID-19 patients have an elevated level of neutrophil cells, which is an unusual situation in other types of viral infection. Neutrophils are also highly related to complications of COVID-19, e.g., thrombosis. In the review by LHA Cavalcante- Silva 48 , there are lots of previous efforts to investigate the neutrophil and COVID-19. The highlight is that COVID-19 triggers the activation of neutrophils. Case study on live cancer staging (Whole blood) SDY1733 data (Oncology, human). This dataset is derived from a liver cancer stage research study 23 . The task is a multi-class classification of the liver cancer stage. The samples are collected from peripheral blood mononuclear cells (PBMC). We derived three live cancer stages, which are hepatic hemangioma (HH), stage A, stage C- untreated, and stage C-treated, based on the Barcelona Clinic Liver Cancer (BCLC) criteria. There are 100 labeled samples. The data is available at SDY1733 on ImmPort. Important marker interpretation. As shown in Figure 5(b) , we list the top 20 important markers and their expression difference in different HCC groups. We can find that the salient markers play important roles in both adaptive and innate immunity. T-cell regulators (CD3, TCR, CD8A, CD28, ICOS) indicated critical information about the T- cell activation and infiltration across different HCC stages. Particularly noteworthy, ICOS is essential in T-cell stimulation and immunotherapy 49 . B-cell markers like CD19 and IGD contribute to humoral immune responses. CD163 and CD172A/B are macrophage- associated molecules, providing insights into tumor-associated macrophages 50 . The identification of immune checkpoints (PD-1, TIM-3) was very important, as they not only indicated the T-cell exhaustion but also provided a target for therapy, e.g., anti-PD-1 immunotherapy. Proliferation and differentiation markers (KI67, TBET, CD127) provided valuable insights about the change of cell populations in HCC prognosis 51 . Notably, CD103 (tissue-resident memory T cells) and CCR7 (lymphocyte homing) suggested distinct stage-specific immune patterns. CD45 (pan-leukocyte markers) and CD11 (integrin family members) are related to the changes of leukocyte composition and cell adhesion dynamics during HCC prognosis. Important cell type interpretation. As shown in Figure 6(b) , we have several observations about the T cell phenotypes. Our analysis revealed distinct CD4+ and CD8+ T cell distribution patterns at different HCC stages. CD4+ T cells were significantly reduced in BCLC-C-treated groups, while CD8+ T cells showed lower proportions in BCLC-A HCC. Notably, total T cells (CD4+ and CD8+) were consistently diminished in both BCLC-A and BCLC-C treated groups compared to other stages, which has also been validated by the study of Shi et al. 23 , suggesting similar immunosuppressive microenvironments in these cohorts. This shared reduction in T cell infiltration may reflect progressive immune dysfunction or treatment-induced modulation, highlighting potential commonalities in immune evasion mechanisms between BCLC-A and treated advanced (HCCtr) HCC. In addition to T cell alterations, we observed elevated monocyte proportions in both BCLC-A and BCLC-C treated groups compared to other HCC stages. This high-level monocyte indicates a potential shift toward myeloid-driven immunosuppression in these cohorts, which aligns with their shared T cell depletion phenotype that has also been validated in the study of Shi et al. 23 . The concurrent rise in monocytes possibly contribute to the tumor-permissive microenvironment in these stages, either through T cell suppression or tumor-associated macrophages polarization. Clinical and scientific insights. Based on the marker and cell levels analysis, we summarized the key insights from clinical and scientific aspects as follows CD172a/b is an important marker in HCC staging classification . From the expression analysis, CD172a/b has significantly higher expression only in the BCLC-C untreated stage. CD172a is an immune suppression receptor that inhibits the phagocytosis of tumor-associated macrophages. This point has also been validated in the previous research on the effectiveness of 3-HAA to HCC 52 . Decreasing the ICOS expression level might be an effective immunotherapy approach for HCC. There is previous work that validates this point. In the study conducted by Lu et al 53 , higher level ICOS+ Tregs are related to worse overall survival. Depleting the ICOS Tregs may be an effective way to improve the clinical outcomes of HCC patients. Because the original study 23 with this data used anti-PD-1 immunotherapy as treatment, this data-driven finding about ICOS may implicitly indicate the mechanism of anti-PD-1 and ICOS expression inhibition. T cells are increased and monocytes are decreased in the BCLC-A groups and the BCLC-C treated groups of HCC. This has been validated by the study that generated this dataset 23 . This finding further underscores the distinct immune landscape of BCLC-A and treated BCLC-C stages. It implicates the myeloid cells as potential and promising therapeutic targets or biomarkers in the HCC prognosis. Case study on HIV diagnosis (PBMC) Important marker interpretation. In Figure 5(c) , the top 20 important markers identified by ImmuneFM are shown. CD4 and CD19 represent central T helper and B cell compartments, always reduced and impaired by HIV. Notably, PD-1 and KI67 reflect T cell proliferation and exhaustion, which can be used as landscapes of chronic HIC infection 54 . HLA-DR, CD54, and CD123 indicate activation of myeloid and dendritic cells. Innate immunity also emerges as essential, with NKG2A, NKp44, KIR2DS4, and ULBP2-5-6 implicating NK cell activation 55 . The presence of HLA-E and HLA-G further indicates changes in non-classical antigen presentation. Markers CD304 and CD112 and chemokine CXCL13 indicate dysregulated lymphoid and immune cell communication 56 . FCRγ is related to antibody-mediated cytotoxicity and phagocytosis, which can be impaired by HIV. Moreover, plasmacytoid dendritic cell markers like CD303 suggest altered interferon responses. As shown in the marker expression difference in Figure 5(c) , the HIV-positive group exhibits significantly lower expression of CD19, CD4, CD54, and CD7, indicating impaired B and T cell function as well as reduced immune adhesion and signaling 57 . On the other hand, increased expression of FCRγ and KI67 reflects heightened immune activation and cell proliferation in response to HIV infection 58 . These expression differences highlight the multifaceted immune dysregulation associated with HIV. Important cell type interpretation. As shown in Figure 6(c) , Natural Killer (NK) cells increased in the HIV-positive groups. This has also been validated in the study of this dataset by Vendrame et al. 26 . This elevation often comes with phenotypic and functional changes, e.g., changed expression of activating and inhibitory receptors and reduced cytotoxicity. Although the proportion is elevated, these NK cells may show signs of exhaustion or dysfunction, which limits their effectiveness and capability to target HIV- infected cells. The alteration of NK cells may be related to immune evasion and provide a promising target for HIV immunotherapy. Additionally, we also observed CD4 T cells decreasing and CD8 T cells increasing in the HIV positive groups. CD4 T cell is the primary target of HIV, causing impaired adaptive immunity. The CD4 loss is accompanied by a relative expansion of CD8 T cells that proliferate in response to antigen exposure. Although the CD8 T cell proportion is elevated, they are often exhausted with markers such as PD-1, TIGIT 26 . The imbalance between CD4 and CD8 T cells reflects ongoing immune dysregulation and activation. The total number of CD4 and CD8 cells has also decreased. Monitoring these alterations provides essential insights into HIV prognosis and potential immunotherapy strategies. Clinical and scientific insights. Based on the marker and cell-level interpretation analysis, we summarize the clinical and scientific insights as follows. CD7 decreases in the HIV infected patients. This point has been validated by previous scientific evidence. Aandahl et al 59 . found that the loss of high- expression CD7 cells is related to HIV infection, especially in patients with fast progression. The reduced CD7 level is caused by the expansion of CD8 T cells with low CD7 expression. KI67 increases in the HIV infected patients. This insight was also validated in previous literature. Sachsenberg et al. 60 measured CD4 and CD8 T cell turnover in HIV patients by KI67. This was also found in our phenotyping results of T cells in Figure 6(c) . The KI67 indicating CD4/CD8 T cells turnover provides promising direction for immunotherapy of HIV. NK cells and CD8 T cells increase, but CD4 T cell decreases in the HIV infected patients. This point was validated in the original study 26 of this HIV downstream data. TIGIT-marked NK cells were elevated in the HIV infected groups. CD4 T cells are known to be the target of HIV, which will significantly decrease after infection. Elevated levels of CD8 T cells may indicate an exhaustion of it. The phenotyping of these immune cells provides promising insights into possible treatment for HIV 61 . Discussion In this work, we curated an AI-ready large-scale cytometry dataset from the ImmPort platform. It is a dataset covering a wide range of immunology-related diseases and conditions. Due to the affordable expense and widespread usage, cytometry data is the most common assay type in immunology. However, various independent studies have different goals and aims. Manual efforts are needed to integrate multi-source datasets. This dataset-level contribution aims to address the lack of a large-scale and comprehensive immunology dataset. It can be used to pre-train the immunology-related foundation model and fine-tune other foundation models. Through building this dataset, 203 commonly used panel markers are included, which can cover most cytometry results analysis. With the Transformer used as the backbone of ImmuneFM, the model with the attention mechanism is able to capture the relation between markers and learn the embeddings of both markers and cells simultaneously. Through the self-supervised learning on a large-scale cytometry dataset, ImmuneFM gains the generalizability across different immunology fields. Compared to the plain Transformer without pre-training, the pre- trained ImmuneFM achieves significantly better performance, which implies the necessity of pre-training on large-scale data. The pre-trained ImmuneFM demonstrates superior performance with a limited number of labeled samples, which is a common challenging scenario in biomedicine. From the quantitative performance results, ImmuneFM outperforms the CNN, Transformer trained from scratch. Moreover, ImmuneFM also outperforms the traditional method based on feature engineering, e.g., FlowSOM+GBDT. It demonstrates the effectiveness of pre-training on large-scale data. Supervised model, including CNN and Transformer, needs enough labeled samples for training, so their performance is not satisfactory with a limited number of labeled cytometry samples. Moreover, the traditional method of feature engineering demonstrates good performance and outperforms supervised deep learning models on some tasks. The feature engineering method, like FlowSOM, can extract the cell distribution information with a limited number of samples, which contributes to the competitive performance. ImmuneFM also accelerates the clinical and scientific discovery in a data-driven way. We analyze three case studies including cancer, HIV, and COVID-19. Through the classification of cytometry samples and the interpretability of ImmuneFM, we can identify the important biomarkers and single cells that contribute to the sample diagnosis. This data-driven method can identify scientific discoveries which has been validated by previous literature and evidence from the wet lab. The immunology foundation model is a good example of how we could use AI for scientific discovery. In the future, we plan to extend this framework to multimodal immunology data, including multi-omics, tabular data, images, and text. Additionally, supporting more immunology- related analytic tasks is a promising direction. Funder Information Declared NIH , HHSN316201200036W References 1. ↵ Marshall , J. S. , Warrington , R. , Watson , W. & Kim , H. L . An introduction to immunology and immunopathology. Allergy , Asthma and Clinical Immunology vol. 14 Preprint at doi: 10.1186/s13223-018-0278-1 ( 2018 ). OpenUrl CrossRef 2. ↵ Spellberg , B. & Edwards , J. E. Type 1/Type 2 Immunity in Infectious Diseases . https://academic.oup.com/cid/article/32/1/76/311106 . 3. ↵ Smith, D. A. & Germolec, D. R. Introduction to Immunology and Autoimmunity . 4. ↵ Finn, O. J. Immuno-oncology: Understanding the function and dysfunction of the immune system in cancer. in Annals of Oncology vol. 23 (Oxford University Press, 2012). 5. ↵ Teijaro , J. R. & Farber , D. L . COVID-19 vaccines: modes of immune activation and future challenges . Nature Reviews Immunology vol. 21 195 – 197 Preprint at doi: 10.1038/s41577-021-00526-x ( 2021 ). OpenUrl CrossRef 6. ↵ Sterner , R. C. & Sterner , R. M . CAR-T cell therapy: current limitations and potential strategies . Blood Cancer Journal vol. 11 Preprint at doi: 10.1038/s41408-021-00459-7 ( 2021 ). OpenUrl CrossRef PubMed 7. ↵ Bhattacharya , S. et al. ImmPort, toward repurposing of open access immunological assay data for translational and clinical research . Sci Data 5 , ( 2018 ). 8. ↵ Gururaj , A. E. , Scheuermann , R. H. & Lin , D . AI and immunology as a new research paradigm . Nat Immunol ( 2024 ) doi: 10.1038/s41590-024-01974-y . OpenUrl CrossRef 9. ↵ Prelaj , A. , et al. Artificial intelligence for predictive biomarker discovery in immuno- oncology: a systematic review . Annals of Oncology vol. 35 29–65 Preprint at doi: 10.1016/j.annonc.2023.10.125 ( 2024 ). 10. ↵ Gangwal , A. et al. Generative artificial intelligence in drug discovery: basic framework, recent advances, challenges, and opportunities. Frontiers in Pharmacology vol. 15 Preprint at doi: 10.3389/fphar.2024.1331062 ( 2024 ). OpenUrl CrossRef PubMed 11. ↵ Zong , N. et al. Computational drug repurposing based on electronic health records: a scoping review. npj Digital Medicine vol . 5 Preprint at doi: 10.1038/s41746-022-00617-6 ( 2022 ). OpenUrl CrossRef PubMed 12. ↵ Bhattacharya, S., Hu, Z. & Butte, A. J. Opportunities and Challenges in Democratizing Immunology Datasets. Frontiers in Immunology vol. 12 Preprint at doi: 10.3389/fimmu.2021.647536 ( 2021 ). 13. ↵ Jiang , T. , Gradus , J. L. & Rosellini , A. J . Supervised Machine Learning: A Brief Primer . Behav Ther 51 , 675 – 687 ( 2020 ). OpenUrl CrossRef PubMed 14. ↵ Uddin , S. , Khan , A. , Hossain , M. E. & Moni , M. A . Comparing different supervised machine learning algorithms for disease prediction . BMC Med Inform Decis Mak 19 , ( 2019 ). 15. ↵ Yakimovich, A., Beaugnon, A., Huang, Y. & Ozkirimli, E. Labels in a haystack: Approaches beyond supervised learning in biomedical applications . Patterns vol. 2 Preprint at doi: 10.1016/j.patter.2021.100383 ( 2021 ). 16. ↵ Liu , X. et al. Biomedical Foundation Model: A Survey . ( 2025 ). 17. ↵ Zhao , W. X. et al. A Survey of Large Language Models . https://www.bing.com/new . 18. ↵ Moor , M. et al. Foundation models for generalist medical artificial intelligence . Nature 616 , 259 – 265 ( 2023 ). OpenUrl CrossRef PubMed 19. ↵ Steinberg , E. , Fries , J. A. , Xu , Y. & Shah , N. H. MOTOR: A TIME-TO-EVENT FOUNDATION MODEL FOR STRUCTURED MEDICAL RECORDS . https://huggingface.co/StanfordShahLab/motor-t-base . 20. ↵ Wang , X. et al. A pathology foundation model for cancer diagnosis and prognosis prediction . Nature ( 2024 ) doi: 10.1038/s41586-024-07894-z . OpenUrl CrossRef 21. ↵ Yang , X. , et al. A large language model for electronic health records. NPJ Digit Med 5, ( 2022 ). 22. ↵ Kim , J. et al. Cytometry masked autoencoder: An accurate and interpretable automated immunophenotyper . Cell Rep Med 5 , 101808 ( 2024 ). 23. ↵ Shi , J. et al. Single-cell immune signature for detecting early-stage HCC and early assessing anti-PD-1 immunotherapy efficacy . J Immunother Cancer 10 , ( 2022 ). 24. ↵ Neeland , M. R. et al. Mass cytometry reveals cellular fingerprint associated with IgE+ peanut tolerance and allergy in early life . Nat Commun 11 , ( 2020 ). 25. ↵ Hoover , P. et al. Accelerating Medicines Partnership: Organizational Structure and Preliminary Data From the Phase 1 Studies of Lupus Nephritis . Arthritis Care Res (Hoboken ) 72 , 233 – 242 ( 2020 ). OpenUrl CrossRef PubMed 26. ↵ Vendrame , E. et al. TIGIT is upregulated by HIV-1 infection and marks a highly functional adaptive and mature subset of natural killer cells . AIDS 34 , 801 – 813 ( 2020 ). OpenUrl CrossRef PubMed 27. ↵ Chen , S. T. et al. A shift in lung macrophage composition is associated with COVID-19 severity and recovery . Sci Transl Med 14 , ( 2022 ). 28. ↵ Wilk , A. J. et al. Multi-omic profiling reveals widespread dysregulation of innate immunity and hematopoiesis in COVID-19 . Journal of Experimental Medicine 218 , ( 2021 ). 29. ↵ Hu , Z. , Tang , A. , Singh , J. , Bhattacharya , S. & Butte , A. J . A robust and interpretable end-to-end deep learning model for cytometry data . PNAS 117 , ( 2003 ). 30. ↵ Yang , Z. et al. XLNet: Generalized Autoregressive Pretraining for Language Understanding . ( 2019 ). 31. ↵ Yang , F. et al. scBERT as a Large-scale Pretrained Deep Language Model for Cell Type Annotation of Single-cell RNA-seq Data . Preprint at doi: 10.1101/2021.12.05.471261 ( 2021 ). OpenUrl Abstract / FREE Full Text 32. ↵ Vaswani , A. , et al. Attention Is All You Need . ( 2017 ). 33. ↵ Neural Network Methods for Natural Language Processing . 34. ↵ Couckuyt , A. , Rombaut , B. , Saeys , Y. & Van Gassen , S . Efficient cytometry analysis with FlowSOM in Python boosts interoperability with other single-cell tools . Bioinformatics 40 , ( 2024 ). 35. Shi , T. et al. Single-cell transcriptomic analysis of renal allograft rejection reveals insights into intragraft TCR clonality . Journal of Clinical Investigation 133 , ( 2023 ). 36. Tursi , A. R. et al. Mass cytometry analysis of blood from peanut-sensitized tolerant and clinically allergic infants . Sci Data 9 , ( 2022 ). 37. Yabu , J. M. , Siebert , J. C. & Maecker , H. T . Immune profiles to predict response to desensitization therapy in highly HLA-sensitized kidney transplant candidates . PLoS One 11 , ( 2016 ). 38. Spitzer , M. H. et al. Systemic Immunity Is Required for Effective Cancer Immunotherapy . Cell 168 , 487 – 502 .e15 ( 2017 ). OpenUrl CrossRef PubMed 39. ↵ Schulte-Schrepping , J. et al. Severe COVID-19 Is Marked by a Dysregulated Myeloid Cell Compartment . Cell 182 , 1419 – 1440 .e23 ( 2020 ). OpenUrl CrossRef PubMed 40. ↵ Morrissey , S. M. et al. A specific low-density neutrophil population correlates with hypercoagulation and disease severity in hospitalized COVID-19 patients . ( 2021 ) doi: 10.1172/jci . OpenUrl CrossRef 41. ↵ Díaz-García , E. et al. Role of CD39 in COVID-19 Severity: Dysregulation of Purinergic Signaling and Thromboinflammation . Front Immunol 13 , ( 2022 ). 42. ↵ Hasan , A. et al. Fatal COVID-19 is Associated with Reduced HLA-DR, CD123 or CD11c Expression on Circulating Dendritic Cells . J Inflamm Res 15 , 5665 – 5675 ( 2022 ). OpenUrl PubMed 43. ↵ Karawajczyk , M. et al. High expression of neutrophil and monocyte CD64 with simultaneous lack of upregulation of adhesion receptors CD11b, CD162, CD15, CD65 on neutrophils in severe COVID-19 . Ther Adv Infect Dis 8 , ( 2021 ). 44. ↵ Ma , Y. , Zhang , Y. & Zhu , L . Role of neutrophils in acute viral infection . Immunity, Inflammation and Disease vol. 9 1186 – 1196 Preprint at doi: 10.1002/iid3.500 ( 2021 ). OpenUrl CrossRef PubMed 45. ↵ da Silva , G. B. et al. High levels of extracellular ATP lead to different inflammatory responses in COVID-19 patients according to the severity . J Mol Med 100 , 645 – 663 ( 2022 ). OpenUrl CrossRef PubMed 46. ↵ Bourgoin , P. et al. CD169 and CD64 could help differentiate bacterial from CoVID- 19 or other viral infections in the Emergency Department . Cytometry Part A 99 , 435 – 445 ( 2021 ). OpenUrl 47. ↵ McKenna , E. et al. Neutrophils in COVID-19: Not Innocent Bystanders . Frontiers in Immunology vol. 13 Preprint at doi: 10.3389/fimmu.2022.864387 ( 2022 ). OpenUrl CrossRef PubMed 48. ↵ Cavalcante-Silva , L. H. A. , et al. Neutrophils and COVID-19: The road so far. International Immunopharmacology vol. 90 Preprint at doi: 10.1016/j.intimp.2020.107233 ( 2021 ). OpenUrl CrossRef PubMed 49. ↵ Dong , C. et al. ICOS co-stimulatory receptor is essential for T-cell activation and function . Nature 409 , 97 – 101 ( 2001 ). OpenUrl CrossRef PubMed Web of Science 50. ↵ Lauwers , Y. et al. Imaging of tumor-associated macrophage dynamics during immunotherapy using a CD163-specific nanobody-based immunotracer . Proceedings of the National Academy of Sciences 121 , ( 2024 ). 51. ↵ Han , J. W. et al. Dynamic Peripheral T-Cell Analysis Identifies On-Treatment Prognostic Biomarkers of Atezolizumab plus Bevacizumab in Hepatocellular Carcinoma . Liver Cancer 1 – 13 ( 2024 ) doi: 10.1159/000541181 . OpenUrl CrossRef 52. ↵ Xue , C. et al. Effects of 3-HAA on HCC by Regulating the Heterogeneous Macrophages—A scRNA-Seq Analysis . Advanced Science 10 , ( 2023 ). 53. ↵ Lu , L.-C. et al. ICOS-Positive Regulatory T Cells in Hepatocellular Carcinoma: The Perspective from Digital Pathology Analysis . Oncology 100 , 419 – 428 ( 2022 ). OpenUrl PubMed 54. ↵ Rueger , S. , et al. Early treatment and PD1 inhibition enhance HIV-specific functionality of follicular CD8+ T cells . JCI Insight 10, ( 2025 ). 55. ↵ Gras Navarro , A., et al. NK Cells with KIR2DS2 Immunogenotype Have a Functional Activation Advantage To Efficiently Kill Glioblastoma and Prolong Animal Survival . The Journal of Immunology 193 , 6192 – 6206 ( 2014 ). OpenUrl PubMed 56. ↵ Rainey-Barger , E. K. et al. The lymphoid chemokine, CXCL13, is dispensable for the initial recruitment of B cells to the acutely inflamed central nervous system . Brain Behav Immun 25 , 922 – 931 ( 2011 ). OpenUrl CrossRef PubMed 57. ↵ Tsitsikov , E. N. , Gutierrez-Ramos , J. C. & Geha , R. S . Impaired CD19 expression and signaling, enhanced antibody response to type II T independent antigen and reduction of B-1 cells in CD81-deficient mice . Proceedings of the National Academy of Sciences 94 , 10844 – 10849 ( 1997 ). OpenUrl Abstract / FREE Full Text 58. ↵ Paiardini , M. & Müller-Trutwin, M. HIV-associated chronic immune activation . Immunol Rev 254 , 78 – 101 ( 2013 ). OpenUrl CrossRef PubMed 59. ↵ Aandahl , E. M . Expansion of CD7low and CD7negative CD8 T-cell effector subsets in HIV-1 infection: correlation with antigenic load and reversion by antiretroviral treatment . Blood 104 , 3672 – 3678 ( 2004 ). OpenUrl Abstract / FREE Full Text 60. ↵ Sachsenberg , N. et al. Turnover of CD4+ and CD8+ T Lymphocytes in HIV-1 Infection as Measured by Ki-67 Antigen . J Exp Med 187 , 1295 – 1303 ( 1998 ). OpenUrl Abstract / FREE Full Text 61. ↵ Collins , D. R. , Gaiha , G. D. & Walker , B. D . CD8+ T cells in HIV control, cure and prevention . Nat Rev Immunol 20 , 471 – 482 ( 2020 ). OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted July 11, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following ImmuneFM: Pre-training Foundation Model from Cytometry Data for Immunology Research Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share ImmuneFM: Pre-training Foundation Model from Cytometry Data for Immunology Research Sirui Ding , Sanchita Bhattacharya , Atul J. Butte bioRxiv 2025.07.09.664020; doi: https://doi.org/10.1101/2025.07.09.664020 Share This Article: Copy Citation Tools ImmuneFM: Pre-training Foundation Model from Cytometry Data for Immunology Research Sirui Ding , Sanchita Bhattacharya , Atul J. Butte bioRxiv 2025.07.09.664020; doi: https://doi.org/10.1101/2025.07.09.664020 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Immunology Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17697) Bioengineering (13895) Bioinformatics (41953) Biophysics (21456) Cancer Biology (18595) Cell Biology (25521) Clinical Trials (138) Developmental Biology (13381) Ecology (19903) Epidemiology (2067) Evolutionary Biology (24323) Genetics (15612) Genomics (22511) Immunology (17738) Microbiology (40401) Molecular Biology (17184) Neuroscience (88623) Paleontology (667) Pathology (2833) Pharmacology and Toxicology (4825) Physiology (7644) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4296) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00