Full text
30,991 characters
· extracted from
preprint-html
· click to expand
CellOntologyMapper: Consensus mapping of cell type annotation | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results CellOntologyMapper: Consensus mapping of cell type annotation View ORCID Profile Zehua Zeng , Xuehai Wang , Hongwu Du doi: https://doi.org/10.1101/2025.06.10.658951 Zehua Zeng a School of Chemistry and Biological Engineering, University of Science and Technology Beijing , Beijing 100083, China b Daxing Research Institute, University of Science and Technology Beijing , Beijing 100083, China c Department of Genetics, Stanford University School of Medicine , Stanford, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Zehua Zeng For correspondence: starlitnightly{at}gmail.com Xuehai Wang d Karolinska Institutet, Department of Learning, Informatics, Management and Ethics , Tomtebodavägen 18 B, 171 65 Solna e Department of Maternal, Child and Adolescent Health, School of Public Health, Anhui Medical University , Hefei, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hongwu Du a School of Chemistry and Biological Engineering, University of Science and Technology Beijing , Beijing 100083, China b Daxing Research Institute, University of Science and Technology Beijing , Beijing 100083, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Single-cell RNA sequencing has revolutionized cellular biology, with atlases now encompassing over 100 million cells. However, researchers employ vastly different naming conventions when annotating cell types, creating a fragmented landscape that severely impedes data integration and comparative analysis. Here, we present CellOntologyMapper, an automated framework that standardizes cell type annotations by intelligently mapping user-defined names to established Cell Ontology and Cell Taxonomy identifiers. Our approach leverages advanced natural language processing, including sentence transformers and large language models, to interpret diverse naming conventions and resolve them to standardized ontological terms. The system handles complex challenges including abbreviated cell names, synonym resolution, and context-dependent interpretation. We built a comprehensive query system based on 19,381 cell type entries from established Cell Ontology and Cell Taxonomy databases, which organize into 24 biologically coherent clusters. Systematic validation across datasets spanning different scales— from 17-cell-type lung studies to 91-cell-type immune atlases and complex developmental systems—consistently demonstrated robust performance and high accuracy. CellOntologyMapper successfully resolved annotation challenges across conventional tissue studies, rare cell populations, and developmentally intricate datasets rich in abbreviated nomenclature. By providing automated, scalable annotation harmonization, our framework enables researchers to leverage existing single-cell datasets while ensuring compatibility with future atlas efforts. When incorporated into publications, CellOntologyMapper enhances the credibility and reference value of cell type annotations, representing a crucial step toward an integrated single-cell genomics ecosystem. Introduction Single-cell RNA sequencing (scRNA-seq) has revolutionized our understanding of cellular diversity across tissues, developmental stages, and disease states. As single-cell atlases continue to expand—with some now encompassing over 100 million cells—accurate cell type annotation has emerged as a fundamental prerequisite for meaningful biological interpretation and cross-study integration [ 1 , 2 ] . Despite this critical importance, the single-cell genomics field faces a persistent challenge: researchers employ vastly different naming conventions when annotating cell types, creating a fragmented landscape that impedes data integration and comparative analysis [ 3 ] . While individual studies may use descriptive names like “inflammatory macrophages,” “CD14+ monocytes,” or simply “Mac1,” these labels often refer to overlapping or identical cell populations, making cross-dataset comparisons extraordinarily difficult [ 4 ] . This challenge becomes even more complex in cross-species studies, where equivalent cell types may be annotated with species-specific nomenclatures that obscure evolutionary relationships and functional conservation patterns [ 5 ] . Current approaches to address this challenge—re-annotating integrated datasets or manual harmonization—introduce substantial limitations [ 6 ] . Re-annotation after data integration risks batch effect confounding, while manual harmonization is labor-intensive, subjective, and scales poorly with expanding data volumes [ 7 ] . The scientific community has developed standardized ontologies, notably the Cell Ontology (CL) and Cell Taxonomy databases, which provide hierarchical, controlled vocabularies for cell type classification [ 8 ] . However, adoption remains inconsistent across the research community, with many studies continuing to use laboratory-specific nomenclatures [ 9 ] . To bridge this critical gap, we developed CellOntologyMapper ( https://github.com/Starlitnightly/CellOntologyMapper ), a computational framework that automatically standardizes cell type annotations by mapping user-defined cell names to established Cell Ontology [ 10 ] and Cell Taxonomy [ 11 ] identifiers. Our approach leverages state-of-the-art natural language processing techniques, including advanced sentence transformers and large language models, to intelligently interpret diverse naming conventions and resolve them to standardized ontological terms. The system handles common challenges including abbreviated cell names, synonym resolution, and context-dependent interpretation based on tissue origin and experimental conditions [ 8 ] . The framework is implemented as an accessible Python package within the OmicVerse [ 12 ] ecosystem, designed to integrate seamlessly into existing single-cell analysis workflows. Result and Discussion Overview of CellOntologyMapper design To address the challenge of inconsistent cell type naming across single-cell studies, we developed CellOntologyMapper, a comprehensive framework that standardizes cell type nomenclature through two interconnected modules ( Figure 1a -b). The first module constructs a robust Cell Ontology Query Database by extracting cell type names from authoritative Cell Ontology and Cell Taxonomy databases along with their associated metadata [ 13 ] . We employed state-of-the-art sentence transformer models, including FlagEmbedding (BAAI) and Qwen (Alibaba), to generate high-dimensional embedding vectors that capture semantic relationships between cell type names [ 14 , 15 ] . These embeddings, coupled with their corresponding cell type identifiers, are efficiently stored in an SQL database optimized for rapid similarity searches ( Figure 1a ). Download figure Open in new tab Figure 1. Overview of CellOntologyMapper for Comprehensive Cell Type Standardization (a) Workflow of constructing the Cell Ontology Query Database. Cell type names were extracted from Cell Ontology and Cell Taxonomy databases, embedding vectors were generated using state-of-the-art sentence transformers (FlagEmbedding, Qwen), and subsequently stored in an SQL database for efficient querying. (b) Pipeline for mapping user-provided cell type abbreviations (e.g., “TA”) to standardized names using regular expression (regex) detection followed by Large Language Model (LLM) amplification. The LLM suggests full cell type names based on sequencing context and tissue background, computes their embedding vectors, ranks similarities against database entries, and selects appropriate Cell Ontology (CL) and Cell Taxonomy (CT) identifiers from top candidates. (c) Summary of the constructed query database containing 19,381 cell type entries (16,841 from Cell Ontology, 2,540 from Cell Taxonomy), visualized on a UMAP embedding to illustrate the distribution of database entries. (d) Unsupervised clustering of cell type embeddings from the query database. Principal component analysis (PCA), neighborhood graph construction, and Leiden clustering identified 24 distinct clusters, each color-coded. (e) Word cloud visualizations highlighting representative cell types from selected clusters. Cluster 7 primarily includes epithelial-related cell types; Cluster 11 is enriched in T cells and subtypes; Cluster 12 consists predominantly of retinal-related cells; Cluster 20 represents fibroblast-related entries; Cluster 21 groups cells associated with liver and pancreas; and Cluster 22 predominantly includes B cells and their subtypes. (f) Example code snippet demonstrating the straightforward and user-friendly API provided by OmicVerse for efficient cell type annotation mapping. The second module implements an intelligent mapping pipeline that transforms user-provided cell type names into standardized ontology terms. To handle the prevalent issue of abbreviated cell type names, we first apply regular expression-based detection to identify potential abbreviations. When abbreviations are detected, we leverage large language models (LLMs) to generate contextually appropriate full cell type names based on experimental metadata such as tissue origin and sequencing protocol [ 16 , 17 ] . The system then computes embedding vectors for the query terms and performs similarity ranking against the entire database. Finally, an LLM-assisted selection process identifies the most appropriate Cell Ontology and Cell Taxonomy identifiers from the top-ranked candidates, ensuring both semantic accuracy and biological relevance ( Figure 1b ). Our comprehensive query database encompasses 19,381 curated cell type entries, comprising 16,841 from Cell Ontology and 2,540 from Cell Taxonomy ( Figure 1c ). To investigate the semantic organization of this database, we performed dimensionality reduction using UMAP and applied unsupervised Leiden clustering, revealing 24 distinct clusters that group semantically related cell types ( Figure 1d ). Remarkably, these clusters exhibit clear biological coherence: cluster 7 predominantly contains epithelial cell types, cluster 11 is enriched for T cells and their specialized subtypes, cluster 12 groups retinal cell populations, cluster 20 encompasses fibroblast-related entries, cluster 21 aggregates hepatic and pancreatic cell types, and cluster 22 primarily consists of B cell lineages and their developmental stages ( Figure 1e ). This clustering pattern validates the biological meaningfulness of our embedding approach and demonstrates the system’s capacity to capture nuanced relationships within the cellular taxonomy. To ensure broad accessibility, we provide a streamlined, user-friendly API through the OmicVerse package that enables researchers to perform cell type standardization with minimal code requirements ( Figure 1f ). This implementation facilitates seamless integration into existing single-cell analysis workflows while maintaining the sophisticated mapping capabilities of the underlying framework. CellOntologyMapper demonstrates robust cross-scale performance across diverse tissue contexts and annotation complexities To rigorously evaluate the versatility and precision of CellOntologyMapper, we conducted comprehensive validation across datasets spanning dramatically different scales and biological contexts, from focused tissue-specific studies to complex developmental systems rich in specialized nomenclature. This systematic assessment encompassed small-scale datasets (50 cell types), and challenging datasets characterized by extensive abbreviations or intricate developmental terminology [ 19 , 20 ] . We first tested CellOntologyMapper on a focused lung tissue dataset (Vieira Braga et al.) containing 17 well-characterized cell types, including immune populations, structural cells, and specialized pulmonary cell lineages [ 21 ] . The system achieved complete accuracy in mapping all annotated cell types to their corresponding Cell Taxonomy identifiers, successfully distinguishing between closely related populations such as alveolar macrophages, interstitial macrophages, and dendritic cell subsets ( Figure 2a ). This performance demonstrates the tool’s precision in handling conventional cell type annotations within established tissue contexts. Download figure Open in new tab Figure 2. Cell Taxonomy-based Annotation of Single-cell Datasets Across Multiple Tissues (a) UMAP visualization of annotated lung single-cell dataset (Vieira Braga et al.) based on Cell Taxonomy. Identified cell types include immune cells (e.g., B cells, macrophages, mast cells), structural cells (basal cells, fibroblasts), and specialized epithelial cells (ciliated, secretory). (b) Annotation of intestinal single-cell data (Haber et al.) visualized by UMAP, highlighting differentiation stages and specialized epithelial populations such as enterocytes, transit amplifying cells (TA), goblet cells, and endocrine cells. (c) UMAP-based annotation of trophoblast single-cell dataset (Arutyunyan et al.) demonstrating diverse placental cell populations including extravillous trophoblasts (EVT), decidual stromal cells (dS), natural killer cells (NK), and syncytiotrophoblasts (SCT), annotated comprehensively via Cell Taxonomy (CT) identifiers. To challenge the system with greater biological complexity, we applied CellOntologyMapper to an intestinal epithelial dataset (Haber et al.) [ 18 ] characterized by extensive cellular differentiation gradients and rare specialized populations. This dataset presented unique annotation challenges, including multiple intermediate differentiation states, region-specific epithelial variants, and highly specialized cell types such as tuft cells and enteroendocrine subtypes. CellOntologyMapper successfully navigated this complexity, accurately annotating diverse populations from proliferative transit-amplifying cells to terminally differentiated absorptive enterocytes, while correctly identifying rare populations like Paneth cells and goblet cell subtypes ( Figure 2b ). This performance highlights the system’s capacity to handle nuanced biological contexts where cell identity exists along continuous differentiation spectra. The most stringent test involved a developmentally complex trophoblast dataset (Arutyunyan et al.) [ 22 ] , which combines the challenges of embryonic cell nomenclature with extensive use of abbreviated cell type names. This dataset encompasses intricate placental development lineages, including syncytiotrophoblasts, cytotrophoblasts, and various extravillous trophoblast populations, many of which are commonly referenced by acronyms in the literature. CellOntologyMapper demonstrated exceptional performance in this challenging context, successfully resolving ambiguous abbreviations and mapping each population to validated Cell Taxonomy entries while maintaining biological coherence across the developmental trajectory ( Figure 2c ). The system’s ability to distinguish between closely related developmental stages, such as different trophoblast progenitor states, underscores its sophisticated understanding of both semantic relationships and biological context. Extended validation across additional datasets of varying scales—including medium-scale bone marrow studies (24 cell types), large-scale bone atlases (55 cell types), and comprehensive immune cell compendiums (91 cell types)—consistently demonstrated high accuracy and computational scalability. These results collectively establish CellOntologyMapper as a robust solution capable of handling the full spectrum of single-cell annotation challenges, from straightforward tissue-specific studies to the most complex developmental and immune system atlases encountered in contemporary single-cell genomics. Supplementary Figure S1 | Cell-Taxonomy-based annotation of human bone-marrow single-cell transcriptomes (Zeng et al . ) (a) UMAP projection of 34,315 bone-marrow cells annotated with Cell Taxonomy (CT) identifiers at a coarse resolution. Distinct clusters correspond to early progenitors (HSC, MPP, CLP, GMP, CMP), lineage-restricted progenitors (MEP, Eo/Baso/Mast precursor, GMP-Neut, GMP-Mono) and mature immune populations including B cells, naïve and memory CD4L/CD8L T cells, NK cells, monocytes, cDCs, pDCs and plasma cells. A small stromal fraction is tentatively labelled “Stromal cell of bone marrow”, yet the transcriptional separation from erythro-myeloid progenitors is modest, cautioning against over-interpretation of this assignment . (b) The identical dataset re-annotated at higher resolution illustrates the hierarchical depth available in the taxonomy. Additional T-cell states emerge—central-memory, effector-memory, tissue-resident and activated subsets—together with discrete stages of B-cell development (immature B, large pre-B, early and late pro-B). Granular erythro-megakaryocytic differentiation (pro-erythroblast, orthochromatic erythroblast, megakaryocyte) and proliferative intermediates (“Cycling progenitor”, “Proliferative T”) are also resolved. Notably, several clusters assigned as “Pre-B” versus “Pro-B cycling” have overlapping marker repertoires, implying that the apparent separation may partly reflect cell-cycle effects rather than true developmental bifurcation . Across both panels, coloured bullets denote CT IDs (CT0000xxxx), providing an explicit, machine-readable linkage to the underlying ontology. While the taxonomy offers a reproducible framework, users should scrutinise clusters with marginal transcriptional distances—especially where cell-cycle or technical variation could masquerade as biological identity . Supplementary Figure S2 | Cell-Taxonomy-based annotation of the cross-tissue human immune-cell atlas (Domínguez Conde et al . ) (a) Uniform Manifold Approximation and Projection (UMAP) of 79 k single cells profiled across 16 tissues and re-labelled with Cell Taxonomy (CT) identifiers. Sixty transcriptionally distinct clusters are recovered. Lymphoid lineages comprise (i) progressive B-cell maturation from immature-B and transitional-B through germinal-centre (GC) B and plasma cells; (ii) a continuum of T-cell states—naïve, central-memory (TCM), effector-memory (TEM), Temra, tissue-resident, follicular-helper (Tfh), regulatory (Treg), γδ-T, mucosal-associated invariant T (MAIT) and NKT cells; and (iii) three innate-lymphoid-cell branches (ILC1–3) in addition to CD56^bright and CD56^dim NK subsets. The myeloid compartment resolves classical, intermediate and non-classical monocytes, neutrophil–myeloid progenitors, conventional dendritic-cell axes (cDC1, cDC2, DC3), LAMP3^+ migratory DCs, macrophage specialisations (Kupffer, kidney-resident, alveolar) and mast cells. Haematopoietic stem and multipotent progenitors (HSC/MPP), lineage-restricted erythro-megakaryocytic (MEP) and granulocyte-monocyte (GMP) precursors, together with sporadic stromal contaminants—endothelial, epithelial and fibroblast signatures—are also detected . Conclusion CellOntologyMapper addresses provide the first comprehensive, automated framework for standardizing cell type nomenclature through intelligent mapping to established ontological databases. Our systematic validation across diverse biological contexts demonstrates robust performance regardless of dataset scale or annotation complexity. The significance of CellOntologyMapper extends beyond technical convenience to enable transformative advances in single-cell biology. By resolving annotation fragmentation, our framework unlocks the potential for large-scale meta-analyses. This capability is particularly crucial as the field constructs comprehensive cellular atlases spanning multiple species, developmental stages, and disease conditions. The tool’s sophisticated handling of abbreviated nomenclatures and biological context ensures compatibility with diverse annotation practices across research communities. Our implementation within the OmicVerse ecosystem provides accessible, user-friendly annotation standardization that will accelerate adoption across the single-cell community. When researchers incorporate CellOntologyMapper results into their publications, their cell type annotations gain enhanced credibility, broader applicability, and improved reference value for the scientific community. As single-cell atlases continue expanding and the volume of cellular data grows exponentially, automated annotation standardization will become essential infrastructure for maintaining scientific rigor and enabling meaningful cross-study comparisons. CellOntologyMapper represents an important step toward a truly integrated single-cell genomics ecosystem that maximizes the scientific value of our collective cellular discoveries. Author Contributions Zehua Zeng : Conceptualization; methodology; software; data curation; investigation; validation; writing—original draft; formal analysis; visualization. Xuehai Wang : methodology; software; data curation; investigation; Hongwu Du : Conceptualization; Writing—review and editing. CONFLICT OF INTEREST STATEMENT The authors declare no conflicts of interest. Footnotes ↵ ⍰ email: zehuazeng{at}xs.ustb.edu.cn ; hongwudu{at}ustb.edu.cn Reference [1]. ↵ Klein D , Palla G , Lange M , et al. Mapping cells through time and space with moscot [J] . Nature , 2025 : 1 – 11 . [2]. ↵ Wang Y , Navin N E. Advances and applications of single-cell sequencing technologies [J] . Molecular cell , 2015 , 58 ( 4 ): 598 – 609 . OpenUrl CrossRef PubMed [3]. ↵ Potter S S. Single-cell RNA sequencing for the study of development, physiology and disease [J] . Nature Reviews Nephrology , 2018 , 14 ( 8 ): 479 – 92 . OpenUrl CrossRef PubMed [4]. ↵ Zhang L , Li Z , Skrzypczynska K M , et al. Single-cell analyses inform mechanisms of myeloid-targeted therapies in colon cancer [J] . Cell , 2020 , 181 ( 2 ): 442 - 59 . e29. OpenUrl CrossRef PubMed [5]. ↵ Song Y , Miao Z , Brazma A , et al. Benchmarking strategies for cross-species integration of single-cell RNA sequencing data [J] . Nature Communications , 2023 , 14 ( 1 ): 6495 . OpenUrl CrossRef PubMed [6]. ↵ Tosti L , Hang Y , Debnath O , et al. Single-nucleus and in situ RNA–sequencing reveal cell topographies in the human pancreas [J] . Gastroenterology , 2021 , 160 ( 4 ): 1330 - 44 . e11. OpenUrl CrossRef PubMed [7]. ↵ Stachelscheid H , Seltmann S , Lekschas F , et al. CellFinder: a cell data repository [J] . Nucleic acids research , 2014 , 42 ( D1 ): D950 – D8 . OpenUrl CrossRef PubMed Web of Science [8]. ↵ Zhang Q , He Y , Luo N , et al. Landscape and dynamics of single immune cells in hepatocellular carcinoma [J] . Cell , 2019 , 179 ( 4 ): 829 - 45 . e20. OpenUrl CrossRef PubMed [9]. ↵ Han X , Zhou Z , Fei L , et al. Construction of a human cell landscape at single-cell level [J] . Nature , 2020 , 581 ( 7808 ): 303 – 9 . OpenUrl CrossRef PubMed [10]. ↵ Diehl A D , Meehan T F , Bradford Y M , et al. The Cell Ontology 2016: enhanced content, modularization, and ontology interoperability [J] . Journal of biomedical semantics , 2016 , 7 : 1 – 10 . OpenUrl CrossRef PubMed [11]. ↵ Jiang S , Qian Q , Zhu T , et al. Cell Taxonomy: a curated repository of cell types with multifaceted characterization [J] . Nucleic Acids Research , 2023 , 51 ( D1 ): D853 – D60 . OpenUrl CrossRef PubMed [12]. ↵ Zeng Z , Ma Y , Hu L , et al. OmicVerse: A single pipeline for exploring the entire transcriptome universe [J] . bioRxiv , 2023 : 2023.06.06.543913. [13]. ↵ Wang S , Pisco A O , Mcgeever A , et al. Leveraging the Cell Ontology to classify unseen cell types [J] . Nature communications , 2021 , 12 ( 1 ): 5556 . OpenUrl CrossRef PubMed [14]. ↵ Yang A , Li A , Yang B , et al. Qwen3 technical report [J] . arXiv preprint arXiv:250509388, 2025 . [15]. ↵ Xiao S , Liu Z , Zhang P , et al. C-pack: Packed resources for general chinese embeddings ; proceedings of the Proceedings of the 47th international ACM SIGIR conference on research and development in information retrieval, F , 2024 [C]. [16]. ↵ Brown T , Mann B , Ryder N , et al. Language models are few-shot learners [J] . Advances in neural information processing systems , 2020 , 33 : 1877 – 901 . OpenUrl [17]. ↵ Thakur N , Reimers N , Daxenberger J , et al. Augmented SBERT: Data augmentation method for improving bi-encoders for pairwise sentence scoring tasks [J] . arXiv preprint arXiv:201008240, 2020 . [18]. ↵ VIEIRA Braga F A, Kar G , Berg M , et al. A cellular census of human lungs identifies novel cell states in health and in asthma [J] . Nature medicine , 2019 , 25 ( 7 ): 1153 – 63 . OpenUrl CrossRef PubMed [19]. ↵ Haber A L , Biton M , Rogel N , et al. A single-cell survey of the small intestinal epithelium [J] . Nature , 2017 , 551 ( 7680 ): 333 – 9 . OpenUrl CrossRef PubMed [20]. ↵ Arutyunyan A , Roberts K , Troulé K , et al. Spatial multiomics map of trophoblast development in early pregnancy [J] . Nature , 2023 , 616 ( 7955 ): 143 – 51 . OpenUrl CrossRef PubMed [21]. ↵ Tasic B , Menon V , Nguyen T N , et al. Adult mouse cortical cell taxonomy revealed by single cell transcriptomics [J] . Nature neuroscience , 2016 , 19 ( 2 ): 335 – 46 . OpenUrl CrossRef PubMed [22]. ↵ Arutyunyan A , Roberts K , Troulé K , et al. Spatial multiomics map of trophoblast development in early pregnancy [J] . Nature , 2023 , 616 ( 7955 ): 143 – 51 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted June 18, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following CellOntologyMapper: Consensus mapping of cell type annotation Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share CellOntologyMapper: Consensus mapping of cell type annotation Zehua Zeng , Xuehai Wang , Hongwu Du bioRxiv 2025.06.10.658951; doi: https://doi.org/10.1101/2025.06.10.658951 Share This Article: Copy Citation Tools CellOntologyMapper: Consensus mapping of cell type annotation Zehua Zeng , Xuehai Wang , Hongwu Du bioRxiv 2025.06.10.658951; doi: https://doi.org/10.1101/2025.06.10.658951 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7619) Biochemistry (17641) Bioengineering (13865) Bioinformatics (41860) Biophysics (21408) Cancer Biology (18545) Cell Biology (25434) Clinical Trials (138) Developmental Biology (13357) Ecology (19863) Epidemiology (2067) Evolutionary Biology (24288) Genetics (15587) Genomics (22466) Immunology (17701) Microbiology (40301) Molecular Biology (17142) Neuroscience (88441) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4814) Physiology (7633) Plant Biology (15108) Scientific Communication and Education (2042) Synthetic Biology (4285) Systems Biology (9811) Zoology (2268)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.