TabVI: Leveraging Lightweight Transformer Architectures to Learn Biologically Meaningful Cellular Representations

preprint OA: closed CC-BY-NC-ND-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Transformer-based foundation models are changing the landscape of natural language processing (NLP), computer vision, and audio, achieving human-level performance across a variety of tasks. Extending these models to single-cell genomics holds significant potential for revealing the cellular and molecular perturbations associated with disease. However, unlike the sequential structure of language, the functional organization of genes is hierarchical and modular. This fundamental difference necessitates the development of meaningful feature selection strategies to adapt NLP transformer architectures effectively. In contrast to many large-scale foundation models, probabilistic models have shown success in learning complex cellular representations from single-cell datasets. In this work, we present TabVI, a probabilistic deep generative model that leverages tabular transformer architectures to improve latent embedding learning. We validate TabVI’s performance in cell type annotation and integration benchmarks. We demonstrate that TabVI improves performance across down-stream tasks and is robust to scaling dataset sizes, producing interpretable, sample-specific feature attention masks. TabVI is a lightweight, scientifically-meaningful, transformer architecture for single-cell analysis that excels where large scale foundation models are less effective.
Full text 65,995 characters · extracted from preprint-html · click to expand
TabVI: Leveraging Lightweight Transformer Architectures to Learn Biologically Meaningful Cellular Representations | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results TabVI: Leveraging Lightweight Transformer Architectures to Learn Biologically Meaningful Cellular Representations Aditi Chandrashekar , Rohan Gala , Andreas Tjärnberg , Saniya Khullar , Grace Huynh , View ORCID Profile Mariano Gabitto doi: https://doi.org/10.1101/2025.02.13.637984 Aditi Chandrashekar 1 California Institute of Technology Find this author on Google Scholar Find this author on PubMed Search for this author on this site Rohan Gala 2 Allen Institute for Brain Science Find this author on Google Scholar Find this author on PubMed Search for this author on this site Andreas Tjärnberg 2 Allen Institute for Brain Science Find this author on Google Scholar Find this author on PubMed Search for this author on this site Saniya Khullar 2 Allen Institute for Brain Science Find this author on Google Scholar Find this author on PubMed Search for this author on this site Grace Huynh 3 Allen Institute Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mariano Gabitto 2 Allen Institute for Brain Science Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Mariano Gabitto For correspondence: mariano.gabitto{at}gmail.com Abstract Full Text Info/History Metrics Preview PDF Abstract Transformer-based foundation models are changing the landscape of natural language processing (NLP), computer vision, and audio, achieving human-level performance across a variety of tasks. Extending these models to single-cell genomics holds significant potential for revealing the cellular and molecular perturbations associated with disease. However, unlike the sequential structure of language, the functional organization of genes is hierarchical and modular. This fundamental difference necessitates the development of meaningful feature selection strategies to adapt NLP transformer architectures effectively. In contrast to many large-scale foundation models, probabilistic models have shown success in learning complex cellular representations from single-cell datasets. In this work, we present TabVI, a probabilistic deep generative model that leverages tabular transformer architectures to improve latent embedding learning. We validate TabVI’s performance in cell type annotation and integration benchmarks. We demonstrate that TabVI improves performance across down-stream tasks and is robust to scaling dataset sizes, producing interpretable, sample-specific feature attention masks. TabVI is a lightweight, scientifically-meaningful, transformer architecture for single-cell analysis that excels where large scale foundation models are less effective. 1 Introduction Large-scale foundation models (FM) Bommasani et al. (2021) pre-trained on vast corpus of data are revolutionizing natural language processing, reaching human-level performance across wide spectrum of applications Devlin et al. (2019) ; OpenAI et al. (2024) ; Dubey et al. (2024) ; Vyas et al. (2023) ; Gardner et al. (2024) ; Oquab et al. (2024) ; Dosovitskiy et al. (2021) ; Wang et al. (2024) . FMs have shown impressive generalization capabilities not only when fine-tuned for downstream tasks, but also when applied in zero-shot paradigms, suggesting the models have captured the underlying structure of language Yin et al. (2019) ; Brown et al. (2020) ; Mercea et al. (2022) ; Li et al. (2024) ; Chen et al. (2024) . Attention-based transformer architectures are at the core of these models Vaswani et al. (2017) . Using such models as a foundation upon which other models can be built is a promising endeavor for scientific machine learning, in which domain-specific scientific problems are solved by using tools from machine learning Subramanian et al. (2023) ; Ho (2024) ; H et al. (2023); Song et al. (2023) . However, it is not known how broadly this methodological approach can be applied or how to readily interpret parameters within transformer architectures or their intermediate layers. While transformer-based architectures and pre-training strategies have been successful for vision and language tasks, the creation and application of foundation models to single cell genomics datasets remains challenging H et al. (2024); Zheng Y (2023); Y et al. (2023); Alsabbagh et al. (2023) ; Boiarsky et al. (2023) ; Kedzierska et al. (2023) . The absence of a clear, sequential ordering of input data (genes), and the noisy and high dimensional nature of -omics data are key challenges in the creation of a single-cell FM. More precisely, NLP transformer architectures incorporate the position of each word within the input to generate context-aware positional embeddings for data processing. While successful when adapted for use in genomics H et al. (2024), genes lack a predefined ordering, rendering this approach less aligned with biological principles. In contrast, variational autoencoder (VAE) Kingma & Welling (2022) -based probabilistic approaches focus on learning unsupervised representations and have been successful in the analysis of large-scale uni- and multi-modal -omics datasets Lopez et al. (2018) ; Ashuach* et al. (2023) ; A. et al. (2022). These models offer a principled approach to describe the data generating process, quantify the uncertainty of estimated quantities, and provide an interpretable description of model parameters, accounting for precise technical and biological factors. VAE based methods are widely used to analyze, annotate, and visualize single-cell RNA-seq (scRNA-seq) data due to their ability to handle high-dimensional, large scale data Lopez et al. (2018) ; Ashuach* et al. (2023) ; A. et al. (2022).VAEs have enabled the construction of a detailed and cohesive understanding of cellular structure within tissues in health and disease Gabitto et al. (2023) . To build such cellular atlases, the heterogeneous patterns of gene expression need to be parsed and cataloged Tjärnberg et al. (2024) , a challenging task given that genes act combinatorially, with a modular and hierarchical interdependence Barabási et al. (2011) , a product of the common developmental origin of cells. In this work, we bridge the gap between the probabilistic and foundation modeling paradigms by developing an approach that incorporates a lightweight transformer module within a VAE representation learning framework for the analysis of scRNA-seq. We harness a recently developed tabular transformer architecture that processes the input data by extracting complex relationships across input features, akin to the one needed in cellular biology given the modular organization of genes, and creating per-sample attention masks Arik & Pfister (2020) . Next, we modify the internal layers of the transformer architecture to enable better sample-efficient performance. We adapt this architecture to work as an encoder network within a probabilistic graphical model, creating a tabular-transformer variational autoencoder. We show that the proposed approach improves representations learned by a commonly-used baseline probabilistic deep generative model A. et al. (2022) and an NLP-based transformer architecture H et al. (2024), and can be learnt in a sample efficient manner, while offering an interpretable view of features and the sample space. These results suggest a path forward for developing scientific foundation models in genomics that leverage biologically-inspired transformer-based architectures to enhance the interpretability of both representations and input features. We showcase the model’s abilities by considering a recent single-nucleus RNA-seq (snRNA-seq) data set from the Seattle Alzheimer’s disease brain Cell Atlas (SEA-AD) Gabitto et al. (2023) , in which individual cells were profiled from heterogeneous brain donors that spanned the entire spectrum of Alzheimer’s disease (AD) states. 2 TabVI TabVI builds upon the probabilistic framework of scVI Lopez et al. (2018) for representation learning. By incorporating both discrete and continuous latent variables, akin to the approach used in scAnVI Xu et al. (2019) , we have developed a model we call TabAnVI. TabAnVI infers a latent cellular space and facilitates the prediction of cell type annotations. In this section, we outline the probabilistic model of TabVI alongside our encoder transformer architecture ( Figure 1 ). We will briefly introduce TabAnVI at the end of this section, details about the model can be found in Appendix A.1. Download figure Open in new tab Figure 1. TabVI model architecture combines the latent space interpretability of VAEs with a sample efficient tabular feature transformer. Input gene expression values pass through feature and attentive transformer layers that provide masks on a per-sample basis. Masked features are encoded by a feature transformer to obtain each cell’s latent low dimensional representation. The representation is decoded through MLP layers to learn parameters of the probabilistic model with a reconstruction objective. 2.1 Probabilistic Model The result of a single cell/nucleus RNA-sequencing (sc/nRNA-seq) experiment generates input data, , depicting the counts of mRNA molecules for the expression of gene g in cell n . Covariates associated with each cell (such as the batch that the cell was collected in) are denoted as s n . We assume that this experiment is generated under observational noise characterized by a zero-inflated negative binomial distribution with parameters θ g , l n , and , denoting gene specific dispersion, cell specific library size, and a mean expression in each cell per gene respectively. This noise model is selected to account for count overdispersion and the inherent sparsity of single cell measurements. Let z n ∈ ℛ d , with d ≪ G (with G being the total number of genes), represent a per-cell variable capturing the underlying cell biological state. The mean expression is calculated by a nonlinear transformation f ψ : ℛ d → ℛ G , with parameters ψ , from the latent cell state. A Bayesian prior is assumed for z n ∈ ℛ K , with K as the dimension of the latent cell space, as a standard normal distribution. This entire model is summarized in equation 1 . To perform inference on our model parameters and due to the intractable nature of their posterior, we resort to variational inference Blei et al. (2017) . In particular, due to the nonlinearities present in the generative model, we use the variational autoencoder framework Kingma & Welling (2022) to perform inference. We assume a variational proposal of the form: To learn the parameters of the model, we optimize the evidence lower bound (ELBO) as a function of the parameters ψ and ϕ using stochastic optimization Kingma & Ba (2017) . In addition, l c and θ g are learnt as point estimates. The functions f ψ and g ϕ are commonly known as decoder and encoder respectively. In our baseline models, scVI and scAnVI models encoders are commonly parameterized by using multi-layer fully connected neural networks. 2.2 Transformer Architecture To improve upon the baseline model and increase the expressiveness of the latent embedding while maintaining interpretability, we adopt a tabular transformer architecture Arik & Pfister (2020) as our encoder. This architecture comprises two primary components: a feature transformer and an attention transformer. The feature transformer maps the input x ∈ ℛ B×G (where B denotes batch size) to an embedding Y ∈ ℛ B× ( D + A ) , which is then split into two parts, Y D and Y A , with dimensions B× D and B × A , respectively. Y D is used to transform the input, while Y A serves as the input for the attention transformer. The attention transformer maps its input Y A ∈ ℛ B×A to ℛ B×G , producing an instance-specific attention mask of the same size as the input data, which is used to re-weight the input. The re-weighted data is then processed by a feature transformer layer. Finally, the output of this layer is passed through a fully connected layer, yielding the posterior mean and variance parameters required for the VAE architecture, as depicted in Figure 1 . Details on the architecture’s building blocks and parameter sizes are provided in Appendix A.2. TabAnVI model is a tabular-transformer extension of the semi-supervised probabilistic algorithm scAnVI. scAnVI provides a principled way to create cell label annotations by enriching the latent structure of scVI with a per-cell categorical variable c n , representing the cell type. Then, each cell has a latent random variable u n , describing biological variability. These two variables are nonlinearly combined into z n , describing a cell type aware state of the cell. Similarly to TabVI and scVI, we use a variational autoencoder framework. Of note, after training, the variational proposals for the categorical variable can be used as a classifier as the latent cell type state can be derived directly from the model q ϕ ( c n | z n ) = Cat( g ϕ ( z n )), with g being a nonlinear function of the latent space with parameters ϕ (for a full description see Appendix A.1). For an ablation study on the components of this architecture, see Appendix A.6. 3 Related Work 4 Empirical Studies 4.1 Cell Type Prediction and Data Integration TabVI aims to learn an interpretable low-dimensional latent space that captures biological variability and corrects for batch and technical effects present in the data. TabAnVI inherits this representation and learns a latent space containing continuous and discrete latent states that can be used for cell type annotation. To evaluate these capabilities, we considered a snRNA-seq data set profiling the human MTG in brain donors with AD from the Seattle Alzheimer’s Disease Brain Cell Atlas (section A.3). Brain donor information can be used as a batch covariate, s n , and condition the model on this information by providing this information for each cell. We randomly selected 80% of the SEA-AD dataset and subset it to the top 4k variable genes as input. We train TabVI, and use this trained model as a pre-trained architecture to train TabAnVI (more on the training procedure in appendix A.4), resulting in a latent representation that can be visualized through UMAP embedding McInnes et al. (2020) ( Figure 2A ). Download figure Open in new tab Figure 2. Benchmarking TabVI model across integration and annotation tasks. A TabVI’s latent representations of human middle temporal gyrus cells originating from SEA-AD donors spanning the entire spectrum of AD. Cells are color-coded by their subclass, where each subclass contains multiple cell types. B, C TabVI representations enable higher overall classification performance compared to scAnVI or scGPT representations, while maintaining scVI’s representation capabilities (as evaluated by using commonly used batch-integration metrics with donors treated as batches). D TabVI achieves high classification performance across cell types. To evaluate latent space properties, we use the scIB platform Luecken et al. (2022) , which provides metrics for assessing dataset integration by measuring batch correction (batch-correction, in our case multiple donors) and the preservation of biological variability (bio-conservation, in our case cell type label). In this evaluation, we examine whether incorporating our transformer architecture affects dataset integration compared to the baseline models, scVI and scAnVI. We find that our enhanced models perform comparably to the parent models ( Figure 2B ). When cell type labels are incorporated into the models (scAnVI and TabAnVI), batch-correction performance remains consistent, while bio-conservation improves significantly, a natural result of training both models with ground truth biological information. We assess the models’ performance in predicting cell types when trained on 4000 genes, using F1 macro annotation scores to measure their ability to classify cell types in the remaining 20% of the dataset, which the models were not trained on. In this extended comparison, we evaluated TabAnVI against scAnVI, scGPT (in zero-shot mode), and scGPT (fine-tuned to our cell type labels) for cell type annotation ( Figure 2C ). TabAnVI outperformed the other models, achieving an F1 score of 0.925, while the next best, scGPT (finetuned), reached 0.857. Notably, both versions of scGPT classified all cell types with varying success (red and gray bars), whereas scVI failed to classify several cell types. TabAnVI showed strong performance across most cell types, missing only one OPC subtype. These computational experiments underscore TabVI’s ability to learn a meaningful latent representation of single-nucleus data and accurately annotate granular cell types, outperforming even a state-of-the-art foundation model. 4.2 Sample and Feature Scaling Properties FM architectures are composed of multiple transformer layers, imposing a high computational burden and requiring vast datasets for training. To study data- and feature-scaling capabilities of our architecture, we build upon our previous computational experiments ( section 4.1 ) and evaluate TabVI’s annotation performance by varying the number of input features (genes) and dataset size. We compare TabVI performance against scAnVI by scaling the number of training samples, N (with N ∈ {50%, 75%, 100%} of the total dataset), and number of genes, G , (with G ∈ {1000, 4000, 10000}), Figure 3 . Download figure Open in new tab Figure 3. Scalability performance evaluated on subsets of features and observations. We compare macro F1 annotation metrics from TabVI against scAnVI when the training data consists of fewer genes, fewer cells, or both. TabVI performance consistently improves when more features are introduced in the input data. Each data point is repeated 3 times; we display mean values. TabVI demonstrates strong performance even when the dataset is subsampled. In contrast, scAnVI under-performs in all settings. Moreover, when the number of input features was expanded from 1,000 to 4,000 genes, TabVI’s performance improved significantly across all data settings. For completeness we also ran the model with 10,000 genes, closer to the order of magnitude available in genomics data, and saw no significant improvement over using the top 4,000 genes. We attribute this improvement in performance to the use of sample-specific transformer attention masks. In contrast, scAnVI either experienced a decline in performance or showed no notable improvement with the increased feature set. In the next section, we will explore the output of attention masks to interpreted their values given the input data. Taken together, these computational experiments demonstrate that the inclusion of a tabular transformer-based architecture in a probabilistic deep generative model lends improvements in the representation of single-cell data sets. Notably, our tabular transformer architecture shows potential as a sample-efficient approach, performing robustly across varying data and feature sizes. 4.3 Meaningful Cell Type-specific Feature Selection To better understand the mechanism by which sample-specific attention masks enhance performance in our tasks, we analyze the attention values produced by the attentive transformer ( Figure 1 ) for the neurons within the Sst subclass, a key population of cortical inhibitory neurons. This subclass is the earliest neuronal population vulnerable to Alzheimer’s Disease pathology Gabitto et al. (2023) ; E. et al. (2022). The Sst subclass contains 16 unique cell types, each containing 1,000 examples in our studied dataset. To better visualize which genes are attended to, we select a threshold to identify the most significant attention values. This threshold is defined as the approximate inflection point in the ranked average attention mask values for each gene across all examples in the Sst subclass ( Figure 4a ). Next, to identify a number of highly attentive genes, we examine the inflection point of the ranked fraction of attentive cells per gene ( Figure 4b ). This thresholding process identifies 200 genes deemed highly attentive in the Sst subclass. Download figure Open in new tab Figure 4. The transformer attention mechanism captures cell type-specific yet combinatorial attention patterns. (a) Line plot depicting the value at the output of the attention transformer, average across all Sst cells (y-axis) for all input genes (x-axis). Values are sorted for visualization purposes. The red line marks the threshold for genes to be considered “active” or “being paid attention to”, set at the inflection point of the attention mask values. (b) Line plot depicting for each gene (x-axis) the fraction of Sst cells in which the gene is active or being paid attention to (y-axis), interpreted as the probability that a gene is attended to, given the input data. Genes are sorted by decreasing attention probability. (c) Heatmap illustrating the attention mask values for all cells within the Sst subclass, arranged hierarchically by genes (x-axis or columns) and stratified by cell types within the Sst subclass (y-axis or rows). With color denoting supertypes within the Sst subclass. TabVI produces highly selective attention masks at both the subclass and supertype levels ( Figure 4c ). These mask-attentive features can be divided into three categories: those that are inactive, those consistently active at the subclass level (left, Figure 4c ), and those that are cell type-specific (middle and right, Figure 4c ). Cell type-specific masks are active across various cell types, reflecting the granular nature of cell type labels, where no single gene uniquely defines a cell type. A closer analysis of gene expression within these masks allows us to distinguish between masks associated with sparsely expressed genes and those linked to genes broadly expressed across cells. Sparsity is a well-known challenge in single-cell datasets G. et al. (2023). Sparsely expressed genes contribute minimally to cell type characterization, as they may be highly selective but are only present in a small number of cells within each type A.5. In contrast, broadly expressed genes exhibit differential expression across cell types and demonstrate strong selectivity across multiple cell types. In summary, these findings show that granular cell type labels are not defined by a single gene but rather by combinations of genes. Cell type annotation algorithms can enhance their accuracy by learning multiple gene patterns in a hierarchical fashion, which vary across subclasses. 5 Discussion Here we introduced TabVI, a deep generative model for the analysis and annotation of single-cell RNA-seq data that harnesses a lightweight and sample efficient tabular transformer architecture to enhance performance on downstream tasks. Using integration and cell type classification benchmarks, we showed that TabVI improves annotation accuracy compared to baseline probabilistic models (scVI/ScAnVI) and NLP-based transformers (scGPT) while retaining latent space interpretability and batch correction capabilities. Future work will focus on studying generalization capabilities of our tabular transformer architecture across vast data sets and out-of-sample examples while extending our model architecture to multimodal single cell assays. A Appendix A.1 TabAnVI Probabilistic Model In the case of TabAnVI, the model is enriched with a discrete cell type label c n , having a multinomial prior, and a hierarchy on the latent biological states u n and z n . The full probabilistic model for TabAnVI is described in equation 4 . where in the previous equation we have overloaded notation to collect all parameters from all nonlinear functions f µ , f σ , and f g into ψ . These functions act as decoders within the variational autoencoder frame-work, mapping from u n to z n and then to . To prevent posterior collapse, a widely studied phenomenon where the posterior of the latent variables is equal to the prior Wang et al. (2023) ; Bowman et al. (2016) ; Chen et al. (2017) , we limit the decoder capacity in TabAnVI and use fully connected networks (similarly to ScAnVI). To infer posterior parameters, we performed approximated Bayesian inference within the variational autoencoder framework. where we have overloaded the notation for all function parameters for g µ , g σ , and g c into variable ϕ . The nonlinear function g µ act as an encoder within the VAE framework as they map input data X g into latent variables z n , and are parametrized with fully connected neural networks in scAnVI and tabular-transformers in TabAnVI. In both models, the network that allows cell type classification is parametrized with fully connected neural network architectures. A.2 Tabular Transformer Architecture In the following description, all fully connected layers FC and gated linear units GLU operate per-sample, while batch normalization BN operates over the samples in the batch as usual. The general feature transformer, f , maps the input x ∈ ℛ B×G to an embedding Y ∈ ℛ B× ( D + A ) . f is composed as neural network blocks FC → BN → GLU with residual connections as shown in Figure 1B . This embedding is split into two parts, Y D and Y A (of dimensions B × D and B × A , respectively). Y D is used to transform the input, while Y A is used as input to the attention transformer in computing attention masks. We set A and D as hyperparameters. The attention transformer, g , maps Y A ∈ R B×A to R B×G , creating mask M . g is composed of blocks FC → BN ( Figure 1C ). Using 1.5-entmax Peters et al. (2019) ensures sparsity in the learned masks, and that ∑ g M bg = 1. The encoder architecture converts input count values, X ∈ ℛ B×G , to normalized and log-transformed values for each sample (cell), X norm ∈ ℛB×G . This input is passed through the feature transformer, f , which outputs an embedding Y 1 . The embedding is split into (discarded in this step) and , which is passed to the attention transformer. The attention transformer g is applied to , producing the attention mask M . This mask is then applied element-wise to the input X norm , producing the masked input . The updated input is passed through the feature transformer f again, resulting in embedding Y 2 . This is split into , which is used for subsequent processing, and , which is discarded as we only consider a single decision step, as in prior methods Fischer et al. (2024) . is passed through a fully connected (FC) layer to generate the posterior mean µ and variance σ 2 as output of the encoder. A.3 Dataset We consider a human middle temporal gyrus snRNA-seq dataset, which consists of 139k cells originating from 84 donors, spanning the entire spectrum of Alzheimer’s disease states Gabitto et al. (2023) . The dataset contains hierarchically organized cellular type annotations, with broad, aggregated subclasses and finer, granular cell types . There are 27 subclasses and 139 cell types. Following standard workflows and to reduce computational costs of benchmarking, we use 1,000, 4,000, and 10,000 highly variable genes in training and analysis, calculated based on gene expression count values, treating donors as the batch variable. The train and test sets are computed by selecting 80% and 20% of the examples in each cell type, respectively. When sub-sampling the dataset (4.2), we select 50% and 70% of the cells in each cell type evenly. In subsequent analysis, dataset size refers to the total number of cells in both train and test sets. A.4 Model training scVI and TabVI are first pre-trained for 500 epochs, then model weights are loaded into their corresponding extensions (scAnVI and TabAnVI) and fine-tuned on labeled data for 35 epochs without freezing the latent space. ScVI and TabVI refer to the pre-trained model, while scAnVI and TabAnVI refer to the fine-tuned model. Cell type annotation metrics are computed using the final fine-tuned model as it is the only one with a cell type annotation layer within its infrastructure. Attention masks are computed through one forward pass of the pre-trained TabVI and thus reflect the model’s attention during reconstruction of the original counts data. All models were trained on NVIDIA A100 Tensor Core GPUs. Model parameters are reported in table 1 . Decision steps refer to the number of iterations in which the attentive architecture is repeated. Attention shared layers refers to the number of shared, fully connected layers in the encoder. Attention independent layers refers to the number of independent GLU layers. The decoder is a multi-layer perceptron (MLP) with its own set of FC layers. View this table: View inline View popup Download powerpoint Table 1: Model Parameters A.5 Cell Type-specific Feature Selection To better understand the selectivity of our attention mechanism, we separate the 200 most active genes into sparsely expressed and non-sparsely expressed sets. Once again, we consider the Sst subclass for analysis. Expression values are first scaled to 10,000 counts per-example, then log1p transformed. To make the distinction between sparse and non-sparse genes, we normalize the expression values per-gene, then sum the expression values per-gene, taking the top 1% of genes as non-sparse and the remaining genes as sparse. This yields 72 non-sparse genes and 128 sparse genes. The expression values in both cases are highly selective of cell type, though this pattern is much more visible in the non-sparse expression map ( Figure 5 ). Download figure Open in new tab Figure 5. Transformer attention masks select combinatorial, highly selective genes. (a) Heatmap illustrating attention masks corresponding to non-sparsely expressed genes for all cells within the Sst subclass, arranged hierarchically by genes (columns) in the mask and hierarchically stratified by cell types within the Sst subclass (rows). (b) Heatmap illustrating normalized expression values corresponding to sparsely expressed genes. Cells and genes organized as in a. (c) Two representative non-sparse genes with their corresponding masks display as a line plot. A.6 Studying Encoder Architecture - Ablation Experiments To better understand the impact of the FT and AT blocks within the network, we conduct a single ablation experiment. We compare the performance of TabAnVI when using two encoder configurations, the first one termed TabAnVI, uses the entire architecture (section A.1). The second one, termed TabAnVI-Attention only, in which only the attention transformer layer is part of the architecture and not the second feature transformer. Parameters for this model are included in table 1 . We compare models trained on 1k, 4k, and 10k genes across 50%, 75%, and 100% of the 139,000 cell SEA-AD dataset. We measure performance using mean F1 score. We show the performance of TabAnVI, TabAnVI AT only, and scAnVI in figure 6 . Download figure Open in new tab Figure 6. All Components of our attention architecture are necessary to increase annotation performance. Performance comparison of TabAnVI, TabANVI Attention only, and scAnVI across multiple data and feature settings. Models were trained on subsets (50%, 75%, and 100%) of the SEA-AD dataset and 1k, 4k, and 10k genes. The ablation study illustrates the impact of the FT and AT blocks within TabAnVI’s architecture Footnotes ajchandr{at}caltech.edu rohang{at}alleninstitute.org andreas.tjaernberg{at}alleninstitute.org saniya.khullar{at}alleninstitute.org grace.huynh{at}alleninstitute.org mariano.gabitto{at}alleninstitute.org References [1]. Gayoso A. , Lopez R. , Xing G. , Boyeau P. , V. Valiollah Pour Amiri , Hong J. , Wu K. , Jayasuriya M. , Mehlman E. , Langevin M. , Liu Y. , Samaran J. , Misrachi G. , Nazaret A. , Clivio O. , Xu C. , Ashuach T. , Gabitto M. , Lotfollahi M. , Svensson V. , and … Yosef N. A python library for probabilistic analysis of single-cell omics data . Nature biotechnology , 40 ( 2 ): 163 – 167 , 2022 . OpenUrl CrossRef PubMed [2]. ↵ Abdel Rahman Alsabbagh , Alberto Maillo Ruiz de Infante , David Gomez-Cabrero , Narsis A. Kiani , Sumeer Ahmad Khan , and Jesper N. Tegnér . Foundation models meet imbalanced single-cell data when learning cell type annotations . bioRxiv , 2023 . doi: 10.1101/2023.10.24.563625 . URL https://www.biorxiv.org/content/early/2023/10/27/2023.10.24.563625 . OpenUrl Abstract / FREE Full Text [3]. ↵ Sercan O. Arik and Tomas Pfister . Tabnet: Attentive interpretable tabular learning , 2020 . URL https://arxiv.org/abs/1908.07442 . [4]. ↵ Tal Ashuach *, Mariano I. Gabitto *, Rohan V. Koodli , Giuseppe-Antonio Saldi , Michael I. Jordan , and Nir Yosef . Multivi: deep generative model for the integration of multimodal data . Nature Methods , 20 ( 8 ): 1222 – 1231 , June 2023 . ISSN 1548-7105 . doi: 10.1038/s41592-023-01909-9 . URL http://dx.doi.org/10.1038/s41592-023-01909-9. OpenUrl CrossRef [5]. ↵ Albert-László Barabási , Natali Gulbahce , and Joseph Loscalzo . Network medicine: a network-based approach to human disease . Nat Rev Genet , 12 ( 1 ): 56 – 68 , January 2011 . OpenUrl CrossRef PubMed Web of Science [6]. ↵ David M. Blei , Alp Kucukelbir , and Jon D. McAuliffe . Variational inference: A review for statisticians . Journal of the American Statistical Association , 112 ( 518 ): 859 – 877 , April 2017 . ISSN 1537-274X . doi: 10.1080/01621459.2017.1285773 . URL http://dx.doi.org/10.1080/01621459.2017.1285773. OpenUrl CrossRef [7]. ↵ Rebecca Boiarsky , Nalini Singh , Alejandro Buendia , Gad Getz , and David Sontag . A deep dive into single-cell rna sequencing foundation models . bioRxiv , 2023 . doi: 10.1101/2023.10.19.563100 . URL https://www.biorxiv.org/content/early/2023/10/23/2023.10.19.563100 . OpenUrl Abstract / FREE Full Text [8]. ↵ Rishi Bommasani , Drew A Hudson , Ehsan Adeli , Russ Altman , Simran Arora , Sydney von Arx , Michael S Bernstein , Jeannette Bohg , Antoine Bosselut , Emma Brunskill , et al. On the opportunities and risks of foundation models . arXiv preprint arxiv: 2108.07258 , 2021 . [9]. ↵ Samuel R. Bowman , Luke Vilnis , Oriol Vinyals , Andrew M. Dai , Rafal Jozefowicz , and Samy Bengio . Generating sentences from a continuous space , 2016 . URL https://arxiv.org/abs/1511.06349 . [10]. ↵ H. Larochelle , M. Ranzato , R. Hadsell , M.F. Balcan , and H. Lin Tom Brown , Benjamin Mann , Nick Ryder , Melanie Subbiah , Jared D Kaplan , Prafulla Dhariwal , Arvind Neelakantan , Pranav Shyam , Girish Sastry , Amanda Askell , Sandhini Agarwal , Ariel Herbert-Voss , Gretchen Krueger , Tom Henighan , Rewon Child , Aditya Ramesh , Daniel Ziegler , Jeffrey Wu , Clemens Winter , Chris Hesse , Mark Chen , Eric Sigler , Mateusz Litwin , Scott Gray , Benjamin Chess , Jack Clark , Christopher Berner , Sam McCandlish , Alec Radford , Ilya Sutskever , and Dario Amodei . Language models are few-shot learners . In H. Larochelle , M. Ranzato , R. Hadsell , M.F. Balcan , and H. Lin (eds.), Advances in Neural Information Processing Systems , volume 33 , pp. 1877 – 1901 . Curran Associates, Inc ., 2020 . URL https://proceedings.neurips.cc/paper_files/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf . OpenUrl [11]. ↵ Shiming Chen , Wenjin Hou , Salman Khan , and Fahad Shahbaz Khan . Progressive semantic-guided vision transformer for zero-shot learning , 2024 . URL https://arxiv.org/abs/2404.07713 . [12]. ↵ Xi Chen , Diederik P. Kingma , Tim Salimans , Yan Duan , Prafulla Dhariwal , John Schulman , Ilya Sutskever , and Pieter Abbeel . Variational lossy autoencoder , 2017 . URL https://arxiv.org/abs/1611.02731 . [13]. ↵ Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . Bert: Pre-training of deep bidirectional transformers for language understanding , 2019 . URL https://arxiv.org/abs/1810.04805 . [14]. ↵ Alexey Dosovitskiy , Lucas Beyer , Alexander Kolesnikov , Dirk Weissenborn , Xiaohua Zhai , Thomas Unterthiner , Mostafa Dehghani , Matthias Minderer , Georg Heigold , Sylvain Gelly , Jakob Uszkoreit , and Neil Houlsby . An image is worth 16×16 words: Transformers for image recognition at scale , 2021 . URL https://arxiv.org/abs/2010.11929 . [15]. ↵ Abhimanyu Dubey , Abhinav Jauhri , Abhinav Pandey , Abhishek Kadian , Ahmad Al-Dahle , Aiesha Letman , Akhil Mathur , Alan Schelten , Amy Yang , Angela Fan , Anirudh Goyal , Anthony Hartshorn , Aobo Yang , Archi Mitra , Archie Sravankumar , Artem Korenev , Arthur Hinsvark , Arun Rao , Aston Zhang , Aurelien Rodriguez , Austen Gregerson , Ava Spataru , Baptiste Roziere , Bethany Biron , Binh Tang , Bobbie Chern , Charlotte Caucheteux , Chaya Nayak , Chloe Bi , Chris Marra , Chris McConnell , Christian Keller , Christophe Touret , Chunyang Wu , Corinne Wong , Cristian Canton Ferrer , Cyrus Nikolaidis , Damien Allonsius , Daniel Song , Danielle Pintz , Danny Livshits , David Esiobu , Dhruv Choudhary , Dhruv Mahajan , Diego Garcia-Olano , Diego Perino , Dieuwke Hupkes , Egor Lakomkin , Ehab AlBadawy , Elina Lobanova , Emily Dinan , Eric Michael Smith , Filip Radenovic , Frank Zhang , Gabriel Synnaeve , Gabrielle Lee , Georgia Lewis Anderson , Graeme Nail , Gregoire Mialon , Guan Pang , Guillem Cucurell , Hailey Nguyen , Hannah Korevaar , Hu Xu , Hugo Touvron , Iliyan Zarov , Imanol Arrieta Ibarra , Isabel Kloumann , Ishan Misra , Ivan Evtimov , Jade Copet , Jaewon Lee , Jan Geffert , Jana Vranes , Jason Park , Jay Mahadeokar , Jeet Shah , Jelmer van der Linde , Jennifer Billock , Jenny Hong , Jenya Lee , Jeremy Fu , Jianfeng Chi , Jianyu Huang , Jiawen Liu , Jie Wang , Jiecao Yu , Joanna Bitton , Joe Spisak , Jongsoo Park , Joseph Rocca , Joshua Johnstun , Joshua Saxe , Junteng Jia , Kalyan Vasuden Alwala , Kartikeya Upasani , Kate Plawiak , Ke Li , Kenneth Heafield , Kevin Stone , Khalid El-Arini , Krithika Iyer , Kshitiz Malik , Kuenley Chiu , Kunal Bhalla , Lauren Rantala-Yeary , Laurens van der Maaten , Lawrence Chen , Liang Tan , Liz Jenkins , Louis Martin , Lovish Madaan , Lubo Malo , Lukas Blecher , Lukas Landzaat , Luke de Oliveira , Madeline Muzzi , Mahesh Pasupuleti , Mannat Singh , Manohar Paluri , Marcin Kardas , Mathew Oldham , Mathieu Rita , Maya Pavlova , Melanie Kambadur , Mike Lewis , Min Si , Mitesh Kumar Singh , Mona Hassan , Naman Goyal , Narjes Torabi , Nikolay Bashlykov , Nikolay Bogoychev , Niladri Chatterji , Olivier Duchenne , Onur Çelebi , Patrick Alrassy , Pengchuan Zhang , Pengwei Li , Petar Vasic , Peter Weng , Prajjwal Bhargava , Pratik Dubal , Praveen Krishnan , Punit Singh Koura , Puxin Xu , Qing He , Qingxiao Dong , Ragavan Srinivasan , Raj Ganapathy , Ramon Calderer , Ricardo Silveira Cabral , Robert Stojnic , Roberta Raileanu , Rohit Girdhar , Rohit Patel , Romain Sauvestre , Ronnie Polidoro , Roshan Sumbaly , Ross Taylor , Ruan Silva , Rui Hou , Rui Wang , Saghar Hosseini , Sahana Chennabasappa , Sanjay Singh , Sean Bell , Seohyun Sonia Kim , Sergey Edunov , Shaoliang Nie , Sharan Narang , Sharath Raparthy , Sheng Shen , Shengye Wan , Shruti Bhosale , Shun Zhang , Simon Vandenhende , Soumya Batra , Spencer Whitman , Sten Sootla , Stephane Collot , Suchin Gururangan , Sydney Borodinsky , Tamar Herman , Tara Fowler , Tarek Sheasha , Thomas Georgiou , Thomas Scialom , Tobias Speckbacher , Todor Mihaylov , Tong Xiao , Ujjwal Karn , Vedanuj Goswami , Vibhor Gupta , Vignesh Ramanathan , Viktor Kerkez , Vincent Gonguet , Virginie Do , Vish Vogeti , Vladan Petrovic , Weiwei Chu , Wenhan Xiong , Wenyin Fu , Whitney Meers , Xavier Martinet , Xiaodong Wang , Xiaoqing Ellen Tan , Xinfeng Xie , Xuchao Jia , Xuewei Wang , Yaelle Goldschlag , Yashesh Gaur , Yasmine Babaei , Yi Wen , Yiwen Song , Yuchen Zhang , Yue Li , Yuning Mao , Zacharie Delpierre Coudert , Zheng Yan , Zhengxing Chen , Zoe Papakipos , Aaditya Singh , Aaron Grattafiori , Abha Jain , Adam Kelsey , Adam Shajnfeld , Adithya Gangidi , Adolfo Victoria , Ahuva Goldstand , Ajay Menon , Ajay Sharma , Alex Boesenberg , Alex Vaughan , Alexei Baevski , Allie Feinstein , Amanda Kallet , Amit Sangani , Anam Yunus , Andrei Lupu , Andres Alvarado , Andrew Caples , Andrew Gu , Andrew Ho , Andrew Poulton , Andrew Ryan , Ankit Ramchandani , Annie Franco , Aparajita Saraf , Arkabandhu Chowdhury , Ashley Gabriel , Ashwin Bharambe , Assaf Eisenman , Azadeh Yazdan , Beau James , Ben Maurer , Benjamin Leonhardi , Bernie Huang , Beth Loyd , Beto De Paola , Bhargavi Paranjape , Bing Liu , Bo Wu , Boyu Ni , Braden Hancock , Bram Wasti , Brandon Spence , Brani Stojkovic , Brian Gamido , Britt Montalvo , Carl Parker , Carly Burton , Catalina Mejia , Changhan Wang , Changkyu Kim , Chao Zhou , Chester Hu , Ching-Hsiang Chu , Chris Cai , Chris Tindal , Christoph Feichtenhofer , Damon Civin , Dana Beaty , Daniel Kreymer , Daniel Li , Danny Wyatt , David Adkins , David Xu , Davide Testuggine , Delia David , Devi Parikh , Diana Liskovich , Didem Foss , Dingkang Wang , Duc Le , Dustin Holland , Edward Dowling , Eissa Jamil , Elaine Montgomery , Eleonora Presani , Emily Hahn , Emily Wood , Erik Brinkman , Esteban Arcaute , Evan Dunbar , Evan Smothers , Fei Sun , Felix Kreuk , Feng Tian , Firat Ozgenel , Francesco Caggioni , Francisco Guzmán , Frank Kanayet , Frank Seide , Gabriela Medina Florez , Gabriella Schwarz , Gada Badeer , Georgia Swee , Gil Halpern , Govind Thattai , Grant Herman , Grigory Sizov , Guangyi , Zhang , Guna Lakshminarayanan , Hamid Shojanazeri , Han Zou , Hannah Wang , Hanwen Zha , Haroun Habeeb , Harrison Rudolph , Helen Suk , Henry Aspegren , Hunter Goldman , Ibrahim Damlaj , Igor Molybog , Igor Tufanov , Irina-Elena Veliche , Itai Gat , Jake Weissman , James Geboski , James Kohli , Japhet Asher , Jean-Baptiste Gaya , Jeff Marcus , Jeff Tang , Jennifer Chan , Jenny Zhen , Jeremy Reizenstein , Jeremy Teboul , Jessica Zhong , Jian Jin , Jingyi Yang , Joe Cummings , Jon Carvill , Jon Shepard , Jonathan McPhie , Jonathan Torres , Josh Ginsburg , Junjie Wang , Kai Wu , Kam Hou U , Karan Saxena , Karthik Prasad , Kartikay Khandelwal , Katayoun Zand , Kathy Matosich , Kaushik Veeraraghavan , Kelly Michelena , Keqian Li , Kun Huang , Kunal Chawla , Kushal Lakhotia , Kyle Huang , Lailin Chen , Lakshya Garg , Lavender A , Leandro Silva , Lee Bell , Lei Zhang , Liangpeng Guo , Licheng Yu , Liron Moshkovich , Luca Wehrstedt , Madian Khabsa , Manav Avalani , Manish Bhatt , Maria Tsimpoukelli , Martynas Mankus , Matan Hasson , Matthew Lennie , Matthias Reso , Maxim Groshev , Maxim Naumov , Maya Lathi , Meghan Keneally , Michael L. Seltzer , Michal Valko , Michelle Restrepo , Mihir Patel , Mik Vyatskov , Mikayel Samvelyan , Mike Clark , Mike Macey , Mike Wang , Miquel Jubert Hermoso , Mo Metanat , Mohammad Rastegari , Munish Bansal , Nandhini Santhanam , Natascha Parks , Natasha White , Navyata Bawa , Nayan Singhal , Nick Egebo , Nicolas Usunier , Nikolay Pavlovich Laptev , Ning Dong , Ning Zhang , Norman Cheng , Oleg Chernoguz , Olivia Hart , Omkar Salpekar , Ozlem Kalinli , Parkin Kent , Parth Parekh , Paul Saab , Pavan Balaji , Pedro Rittner , Philip Bontrager , Pierre Roux , Piotr Dollar , Polina Zvyagina , Prashant Ratanchandani , Pritish Yuvraj , Qian Liang , Rachad Alao , Rachel Rodriguez , Rafi Ayub , Raghotham Murthy , Raghu Nayani , Rahul Mitra , Raymond Li , Rebekkah Hogan , Robin Battey , Rocky Wang , Rohan Maheswari , Russ Howes , Ruty Rinott , Sai Jayesh Bondu , Samyak Datta , Sara Chugh , Sara Hunt , Sargun Dhillon , Sasha Sidorov , Satadru Pan , Saurabh Verma , Seiji Yamamoto , Sharadh Ramaswamy , Shaun Lindsay , Shaun Lindsay , Sheng Feng , Shenghao Lin , Shengxin Cindy Zha , Shiva Shankar , Shuqiang Zhang , Shuqiang Zhang , Sinong Wang , Sneha Agarwal , Soji Sajuyigbe , Soumith Chintala , Stephanie Max , Stephen Chen , Steve Kehoe , Steve Satterfield , Sudarshan Govindaprasad , Sumit Gupta , Sungmin Cho , Sunny Virk , Suraj Subramanian , Sy Choudhury , Sydney Goldman , Tal Remez , Tamar Glaser , Tamara Best , Thilo Kohler , Thomas Robinson , Tianhe Li , Tianjun Zhang , Tim Matthews , Timothy Chou , Tzook Shaked , Varun Vontimitta , Victoria Ajayi , Victoria Montanez , Vijai Mohan , Vinay Satish Kumar , Vishal Mangla , Vítor Albiero , Vlad Ionescu , Vlad Poenaru , Vlad Tiberiu Mihailescu , Vladimir Ivanov , Wei Li , Wenchen Wang , Wenwen Jiang , Wes Bouaziz , Will Constable , Xiaocheng Tang , Xiaofang Wang , Xiaojian Wu , Xiaolan Wang , Xide Xia , Xilun Wu , Xinbo Gao , Yanjun Chen , Ye Hu , Ye Jia , Ye Qi , Yenda Li , Yilin Zhang , Ying Zhang , Yossi Adi , Youngjin Nam , Yu , Wang , Yuchen Hao , Yundi Qian , Yuzi He , Zach Rait , Zachary DeVito , Zef Rosnbrick , Zhaoduo Wen , Zhenyu Yang , and Zhiwei Zhao . The llama 3 herd of models , 2024 . [16]. Consens M. E. , Chen Y. , Menon V. , Wang Y. , Schneider J. A. , De Jager Philip L. , Bennett D. A. , Tripathy S. J. , and Felsky D. Bulk and single-nucleus transcriptomics highlight intra-telencephalic and somatostatin neurons in alzheimer’s disease . Frontiers in Molecular Neuroscience , 15 , 2022 . [17]. ↵ Felix Fischer , David S Fischer , Roman Mukhin , Andrey Isaev , Evan Biederstedt , Alexandra-Chloé Villani and Fabian J Theis . sctab: Scaling cross-tissue single-cell annotation models . Nature Communications , 15 ( 1 ): 6611 , 2024 . OpenUrl CrossRef PubMed [18]. Bouland G. , Mahfouz A. , and Reinders M. J. T. Consequences and opportunities arising due to sparser single-cell rna-seq datasets , 2023 . [19]. ↵ Mariano I. Gabitto , Kyle J. Travaglini , Victoria M. Rachleff , Eitan S. Kaplan , Brian Long , Jeanelle Ariza , Yi Ding , Joseph T. Mahoney , Nick Dee , Jeff Goldy , Erica J. Melief , Krissy Brouner , Jazmin Campos , John Campos , Ambrose J. Carr , Tamara Casper , Rushil Chakrabarty , Michael Clark , Jonah Cool , Nasmil J. Valera Cuevas , Rachel Dalley , Martin Darvas , Song-Lin Ding , Tim Dolbeare , Christine L. Mac Donald , Tom Egdorf , Luke Esposito , Rebecca Ferrer , Rohan Gala , Amanda Gary , Jessica Gloe , Nathan Guilford , Junitta Guzman , Daniel Hirschstein , Windy Ho , Tim Jarksy , Nelson Johansen , Brian E. Kalmbach , Lisa M. Keene , Sarah Khawand , Mitch Kilgore , Amanda Kirkland , Michael Kunst , Brian R. Lee , Jocelin Malone , Zoe Maltzer , Naomi Martin , Rachel McCue , Delissa McMillen , Emma Meyerdierks , Kelly P. Meyers , Tyler Mollenkopf , Mark Montine , Amber L. Nolan , Julie Nyhus , Paul A. Olsen , Maiya Pacleb , Nicholas Peña , Thanh Pham , Christina Alice Pom , Nadia Postupna , Augustin Ruiz , Aimee M. Schantz , Nadiya V. Shapovalova , Staci A. Sorensen , Brian Staats , Matt Sullivan , Susan M. Sunkin , Carol Thompson , Michael Tieu , Jonathan Ting , Amy Torkelson , Tracy Tran , Ming-Qiang Wang , Jack Waters , Angela M. Wilson , David Haynor , Nicole Gatto , Suman Jayadev , Shoaib Mufti , Lydia Ng , Shubhabrata Mukherjee , Paul K. Crane , Caitlin S. Latimer , Boaz P. Levi , Kimberly Smith , Jennie L. Close , Jeremy A. Miller , Rebecca D. Hodge , Eric B. Larson , Thomas J. Grabowski , Michael Hawrylycz , C. Dirk Keene , and Ed S. Lein . Integrated multimodal cell atlas of alzheimer’s disease . May 2023 . doi: 10.1101/2023.05.08.539485 . URL http://dx.doi.org/10.1101/2023.05.08.539485. OpenUrl Abstract / FREE Full Text [20]. ↵ Josh Gardner , Simon Durand , Daniel Stoller , and Rachel M. Bittner . Llark: A multimodal instruction-following language model for music , 2024 . URL https://arxiv.org/abs/2310.07160 . [21]. Cui H , Wang C , Maan H , Pang K , Luo F , Duan N , and Wang B. scgpt: toward building a foundation model for single-cell multi-omics using generative ai . Nature Methods , 21 ( 8 ): 1470 – 1480 , 2024 . OpenUrl CrossRef PubMed [22]. ↵ Wang H , Fu T , Du Y , Gao W , Huang K , Liu Z , Chandak P , Liu S , Van Katwyk P , Deac A , Anandkumar A , Bergen K , Gomes CP , Ho S , Kohli P , Lasenby J , Leskovec J , Manrai A Liu TY , Marks D , Ramsundar B , Song L , Sun J , Tang J , Veličković P , Welling M , Zhang L , Coley CW , Bengio Y , and Zitnik M. Scientific discovery in the age of artificial intelligence . Nature , 621 ( 7972 ): 47 – 60 , 2023 . OpenUrl CrossRef PubMed [23]. ↵ Shirley Ho . Foundation Models for Science (and Astrophysics!) . In American Astronomical Society Meeting Abstracts, volume 243 of American Astronomical Society Meeting Abstracts , pp. 120 .01, February 2024 . [24]. ↵ Kasia Z. Kedzierska , Lorin Crawford , Ava P. Amini , and Alex X. Lu . Assessing the limits of zero-shot foundation models in single-cell biology . bioRxiv , 2023 . doi: 10.1101/2023.10.16.561085 . URL https://www.biorxiv.org/content/early/2023/10/17/2023.10.16.561085 . OpenUrl Abstract / FREE Full Text [25]. ↵ Diederik P. Kingma and Jimmy Ba . Adam: A method for stochastic optimization , 2017 . URL https://arxiv.org/abs/1412.6980 . [26]. ↵ Diederik P Kingma and Max Welling . Auto-encoding variational bayes , 2022 . URL https://arxiv.org/abs/1312.6114 . [27]. ↵ Wenrui Li , Penghong Wang , Ruiqin Xiong , and Xiaopeng Fan . Spiking tucker fusion transformer for audio-visual zero-shot learning , 2024 . URL https://arxiv.org/abs/2407.08130 . [28]. ↵ Romain Lopez , Jeffrey Regier , Michael B. Cole , Michael I. Jordan , and Nir Yosef . Deep generative modeling for single-cell transcriptomics . Nature Methods , 15 ( 12 ): 1053 – 1058 , November 2018 . ISSN 1548-7105 . doi: 10.1038/s41592-018-0229-2 . URL http://dx.doi.org/10.1038/s41592-018-0229-2. OpenUrl CrossRef PubMed [29]. ↵ Malte D Luecken , Maren Büttner, Kridsadakorn Chaichoompu , Anna Danese , Marta Interlandi , Michaela F Müller , Daniel C Strobl , Luke Zappia , Martin Dugas , Maria Colomé-Tatché , et al. Benchmarking atlas-level data integration in single-cell genomics . Nature methods , 19 ( 1 ): 41 – 50 , 2022 . OpenUrl CrossRef PubMed [30]. ↵ Leland McInnes , John Healy , and James Melville . Umap: Uniform manifold approximation and projection for dimension reduction , 2020 . URL https://arxiv.org/abs/1802.03426 . [31]. ↵ Otniel-Bogdan Mercea , Lukas Riesch , A. Sophia Koepke , and Zeynep Akata . Audio-visual generalised zeroshot learning with cross-modal attention and language , 2022 . URL https://arxiv.org/abs/2203.03598 . [32]. ↵ OpenAI , Josh Achiam , Steven Adler , Sandhini Agarwal , Lama Ahmad , Ilge Akkaya , Florencia Leoni Aleman , Diogo Almeida , Janko Altenschmidt , Sam Altman , Shyamal Anadkat , Red Avila , Igor Babuschkin , Suchir Balaji , Valerie Balcom , Paul Baltescu , Haiming Bao , Mohammad Bavarian , Jeff Belgum , Irwan Bello , Jake Berdine , Gabriel Bernadett-Shapiro , Christopher Berner , Lenny Bogdonoff , Oleg Boiko , Madelaine Boyd , Anna-Luisa Brakman , Greg Brockman , Tim Brooks , Miles Brundage , Kevin Button , Trevor Cai , Rosie Campbell , Andrew Cann , Brittany Carey , Chelsea Carlson , Rory Carmichael , Brooke Chan , Che Chang , Fotis Chantzis , Derek Chen , Sully Chen , Ruby Chen , Jason Chen , Mark Chen , Ben Chess , Chester Cho , Casey Chu , Hyung Won Chung , Dave Cummings , Jeremiah Currier , Yunxing Dai , Cory Decareaux , Thomas Degry , Noah Deutsch , Damien Deville , Arka Dhar , David Dohan , Steve Dowling , Sheila Dunning , Adrien Ecoffet , Atty Eleti , Tyna Eloundou , David Farhi , Liam Fedus , Niko Felix , Simón Posada Fishman , Juston Forte , Isabella Fulford , Leo Gao , Elie Georges , Christian Gibson , Vik Goel , Tarun Gogineni , Gabriel Goh , Rapha Gontijo-Lopes , Jonathan Gordon , Morgan Grafstein , Scott Gray , Ryan Greene , Joshua Gross , Shixiang Shane Gu , Yufei Guo , Chris Hallacy , Jesse Han , Jeff Harris , Yuchen He , Mike Heaton , Johannes Heidecke , Chris Hesse , Alan Hickey , Wade Hickey , Peter Hoeschele , Brandon Houghton , Kenny Hsu , Shengli Hu , Xin Hu , Joost Huizinga , Shantanu Jain , Shawn Jain , Joanne Jang , Angela Jiang , Roger Jiang , Haozhun Jin , Denny Jin , Shino Jomoto , Billie Jonn , Heewoo Jun , Tomer Kaftan , Łukasz Kaiser , Ali Kamali , Ingmar Kanitscheider , Nitish Shirish Keskar , Tabarak Khan , Logan Kilpatrick , Jong Wook Kim , Christina Kim , Yongjik Kim , Jan Hendrik Kirchner , Jamie Kiros , Matt Knight , Daniel Kokotajlo , Łukasz Kondraciuk , Andrew Kondrich , Aris Konstantinidis , Kyle Kosic , Gretchen Krueger , Vishal Kuo , Michael Lampe , Ikai Lan , Teddy Lee , Jan Leike , Jade Leung , Daniel Levy , Chak Ming Li , Rachel Lim , Molly Lin , Stephanie Lin , Mateusz Litwin , Theresa Lopez , Ryan Lowe , Patricia Lue , Anna Makanju , Kim Malfacini , Sam Manning , Todor Markov , Yaniv Markovski , Bianca Martin , Katie Mayer , Andrew Mayne , Bob McGrew , Scott Mayer McKinney , Christine McLeavey , Paul McMillan , Jake McNeil , David Medina , Aalok Mehta , Jacob Menick , Luke Metz , Andrey Mishchenko , Pamela Mishkin , Vinnie Monaco , Evan Morikawa , Daniel Mossing , Tong Mu , Mira Murati , Oleg Murk , David Mély , Ashvin Nair , Reiichiro Nakano , Rajeev Nayak , Arvind Neelakantan , Richard Ngo , Hyeonwoo Noh , Long Ouyang , Cullen O’Keefe , Jakub Pachocki , Alex Paino , Joe Palermo , Ashley Pantuliano , Giambattista Parascandolo , Joel Parish , Emy Parparita , Alex Passos , Mikhail Pavlov , Andrew Peng , Adam Perelman , Filipe de Avila Belbute Peres , Michael Petrov , Henrique Ponde de Oliveira Pinto , Michael , Pokorny , Michelle Pokrass , Vitchyr H. Pong , Tolly Powell , Alethea Power , Boris Power , Elizabeth Proehl , Raul Puri , Alec Radford , Jack Rae , Aditya Ramesh , Cameron Raymond , Francis Real , Kendra Rimbach , Carl Ross , Bob Rotsted , Henri Roussez , Nick Ryder , Mario Saltarelli , Ted Sanders , Shibani Santurkar , Girish Sastry , Heather Schmidt , David Schnurr , John Schulman , Daniel Selsam , Kyla Sheppard , Toki Sherbakov , Jessica Shieh , Sarah Shoker , Pranav Shyam , Szymon Sidor , Eric Sigler , Maddie Simens , Jordan Sitkin , Katarina Slama , Ian Sohl , Benjamin Sokolowsky , Yang Song , Natalie Staudacher , Felipe Petroski Such , Natalie Summers , Ilya Sutskever , Jie Tang , Nikolas Tezak , Madeleine B. Thompson , Phil Tillet , Amin Tootoonchian , Elizabeth Tseng , Preston Tuggle , Nick Turley , Jerry Tworek , Juan Felipe Cerón Uribe , Andrea Vallone , Arun Vijayvergiya , Chelsea Voss , Carroll Wainwright , Justin Jay Wang , Alvin Wang , Ben Wang , Jonathan Ward , Jason Wei , CJ Weinmann , Akila Welihinda , Peter Welinder , Jiayi Weng , Lilian Weng , Matt Wiethoff , Dave Willner , Clemens Winter , Samuel Wolrich , Hannah Wong , Lauren Workman , Sherwin Wu , Jeff Wu , Michael Wu , Kai Xiao , Tao Xu , Sarah Yoo , Kevin Yu , Qiming Yuan , Wojciech Zaremba , Rowan Zellers , Chong Zhang , Marvin Zhang , Shengjia Zhao , Tianhao Zheng , Juntang Zhuang , William Zhuk , and Barret Zoph . Gpt-4 technical report , 2024 . URL https://arxiv.org/abs/2303.08774 . [33]. ↵ Maxime Oquab , Timothée Darcet , Théo Moutakanni , Huy Vo , Marc Szafraniec , Vasil Khalidov , Pierre Fernandez , Daniel Haziza , Francisco Massa , Alaaeldin El-Nouby , Mahmoud Assran , Nicolas Ballas , Wojciech Galuba , Russell Howes , Po-Yao Huang , Shang-Wen Li , Ishan Misra , Michael Rabbat , Vasu Sharma , Gabriel Synnaeve , Hu Xu , Hervé Jegou Julien Mairal , Patrick Labatut , Armand Joulin , and Piotr Bojanowski . Dinov2: Learning robust visual features without supervision , 2024 . URL https://arxiv.org/abs/2304.07193 . [34]. Ben Peters , Vlad Niculae , and André FT Martins . Sparse sequence-to-sequence models . arXiv preprint arxiv: 1905.05702 , 2019 . [35]. ↵ Shuaiwen Leon Song , Bonnie Kruft , Minjia Zhang , Conglong Li , Shiyang Chen , Chengming Zhang , Masahiro Tanaka , Xiaoxia Wu , Jeff Rasley , Ammar Ahmad Awan , Connor Holmes , Martin Cai , Adam Ghanem , Zhongzhu Zhou , Yuxiong He , Pete Luferenko , Divya Kumar , Jonathan Weyn , Ruixiong Zhang , Sylwester Klocek , Volodymyr Vragov , Mohammed AlQuraishi , Gustaf Ahdritz , Christina Floristean , Cristina Negri , Rao Kotamarthi , Venkatram Vishwanath , Arvind Ramanathan , Sam Foreman , Kyle Hippe , Troy Arcomano , Romit Maulik , Maxim Zvyagin , Alexander Brace , Bin Zhang, Cindy Orozco Bohorquez , Austin Clyde , Bharat Kale , Danilo Perez-Rivera , Heng Ma , Carla M. Mann , Michael Irvin , J. Gregory Pauloski , Logan Ward , Valerie Hayot , Murali Emani , Zhen Xie , Diangen Lin , Maulik Shukla , Ian Foster , James J. Davis , Michael E. Papka , Thomas Brettin , Prasanna Balaprakash , Gina Tourassi , John Gounley , Heidi Hanson , Thomas E Potok , Massimiliano Lupo Pasini , Kate Evans , Dan Lu , Dalton Lunga , Junqi Yin , Sajal Dash , Feiyi Wang , Mallikarjun Shankar , Isaac Lyngaas , Xiao Wang , Guojing Cong , Pei Zhang , Ming Fan , Siyan Liu , Adolfy Hoisie , Shinjae Yoo , Yihui Ren , William Tang , Kyle Felker , Alexey Svyatkovskiy , Hang Liu , Ashwin Aji , Angela Dalton , Michael Schulte , Karl Schulz , Yuntian Deng , Weili Nie , Josh Romero , Christian Dallago , Arash Vahdat , Chaowei Xiao , Thomas Gibbs , Anima Anandkumar , and Rick Stevens . Deepspeed4science initiative: Enabling large-scale scientific discovery through sophisticated ai system technologies , 2023 . URL https://arxiv.org/abs/2310.04610 . [36]. ↵ A. Oh , T. Naumann , A. Globerson , K. Saenko , M. Hardt , and S. Levine Shashank Subramanian , Peter Harrington , Kurt Keutzer , Wahid Bhimji , Dmitriy Morozov , Michael W Mahoney , and Amir Gholami . Towards foundation models for scientific machine learning: Characterizing scaling and transfer behavior . In A. Oh , T. Naumann , A. Globerson , K. Saenko , M. Hardt , and S. Levine (eds.), Advances in Neural Information Processing Systems , volume 36 , pp. 71242 – 71262 . Curran Associates, Inc ., 2023 . URL https://proceedings.neurips.cc/paper_files/paper/2023/file/e15790966a4a9d85d688635c88ee6d8a-Paper-Conference.pdf . OpenUrl [37]. ↵ Andreas Tjärnberg , Maggie Beheler-Amass , Christopher Jackson , Lionel Christiaen , David Gresham , and Richard Bonneau . Structure-primed embedding on the transcription factor manifold enables transparent model architectures for gene regulatory network and latent activity inference . Genome Biology , 25 ( 1 ): 24 , Jan 2024 . ISSN 1474-760X . doi: 10.1186/s13059-023-03134-1 . URL 10.1186/s13059-023-03134-1. OpenUrl CrossRef [38]. ↵ I. Guyon , U. Von Luxburg , S. Bengio , H. Wallach , R. Fergus , S. Vishwanathan , and R. Garnett Ashish Vaswani , Noam Shazeer , Niki Parmar , Jakob Uszkoreit , Llion Jones , Aidan N Gomez , Łukasz Kaiser , and Illia Polosukhin . Attention is all you need . In I. Guyon , U. Von Luxburg , S. Bengio , H. Wallach , R. Fergus , S. Vishwanathan , and R. Garnett (eds.), Advances in Neural Information Processing Systems , volume 30 . Curran Associates, Inc ., 2017 . URL https://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf . [39]. ↵ Apoorv Vyas , Bowen Shi , Matthew Le , Andros Tjandra , Yi-Chiao Wu , Baishan Guo , Jiemin Zhang , Xinyue Zhang , Robert Adkins , William Ngan , Jeff Wang , Ivan Cruz , Bapi Akula , Akinniyi Akinyemi , Brian Ellis , Rashel Moritz , Yael Yungster , Alice Rakotoarison , Liang Tan , Chris Summers , Carleigh Wood , Joshua Lane , Mary Williamson , and Wei-Ning Hsu . Audiobox: Unified audio generation with natural language prompts , 2023 . URL https://arxiv.org/abs/2312.15821 . [40]. ↵ Chien-Yao Wang , I-Hau Yeh , and Hong-Yuan Mark Liao . Yolov9: Learning what you want to learn using programmable gradient information , 2024 . URL https://arxiv.org/abs/2402.13616 . [41]. Yixin Wang , David M. Blei , and John P. Cunningham . Posterior collapse and latent variable nonidentifiability , 2023 . URL https://arxiv.org/abs/2301.00537 . [42]. ↵ Chenling Xu , Romain Lopez , Edouard Mehlman , Jeffrey Regier , Michael I. Jordan , and Nir Yosef . Probabilistic harmonization and annotation of single-cell transcriptomics data with deep generative models . January 2019 . doi: 10.1101/532895 . URL http://dx.doi.org/10.1101/532895. OpenUrl Abstract / FREE Full Text [43]. Rosen Y , Roohani Y , Agarwal A , Samotorčan L , Quake S R , and Leskovec J. Universal cell embeddings: A foundation model for cell biology . bioRxiv , 2023 . doi: 10.1101/2023.11.28.568918 . URL https://www.biorxiv.org/content/early/2023/11/29/2023.11.28.568918 . OpenUrl Abstract / FREE Full Text [44]. ↵ Wenpeng Yin , Jamaal Hay , and Dan Roth . Benchmarking zero-shot text classification: Datasets, evaluation and entailment approach , 2019 . URL https://arxiv.org/abs/1909.00161 . [45]. Gao GF. Zheng Y. Geneformer: a deep learning model for exploring gene networks . Sci China Life Sci ., 66 ( 12 ): 2952 – 2954 , 2023 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted February 17, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following TabVI: Leveraging Lightweight Transformer Architectures to Learn Biologically Meaningful Cellular Representations Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share TabVI: Leveraging Lightweight Transformer Architectures to Learn Biologically Meaningful Cellular Representations Aditi Chandrashekar , Rohan Gala , Andreas Tjärnberg , Saniya Khullar , Grace Huynh , Mariano Gabitto bioRxiv 2025.02.13.637984; doi: https://doi.org/10.1101/2025.02.13.637984 Share This Article: Copy Citation Tools TabVI: Leveraging Lightweight Transformer Architectures to Learn Biologically Meaningful Cellular Representations Aditi Chandrashekar , Rohan Gala , Andreas Tjärnberg , Saniya Khullar , Grace Huynh , Mariano Gabitto bioRxiv 2025.02.13.637984; doi: https://doi.org/10.1101/2025.02.13.637984 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17697) Bioengineering (13894) Bioinformatics (41951) Biophysics (21455) Cancer Biology (18593) Cell Biology (25509) Clinical Trials (138) Developmental Biology (13380) Ecology (19903) Epidemiology (2067) Evolutionary Biology (24322) Genetics (15611) Genomics (22509) Immunology (17737) Microbiology (40398) Molecular Biology (17183) Neuroscience (88619) Paleontology (667) Pathology (2833) Pharmacology and Toxicology (4825) Physiology (7644) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4296) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-26T02:00:01.498150+00:00
License: CC-BY-NC-ND-4.0