Full text
81,332 characters
· extracted from
preprint-html
· click to expand
Identification and quantification of alternative polyadenylation sites in single cell RNA-seq data using scPAISO | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Identification and quantification of alternative polyadenylation sites in single cell RNA-seq data using scPAISO Yongjie Liu , Peiwen Xiong , Songyang Li , Xinjia Liu , Tao Liu , Qinglan Yang , Shuting Wu , Hongyan Peng , Yana Li , Lingling Zhang , Yafei Deng , Yong Zhu , Junping Wang , Youcai Deng doi: https://doi.org/10.1101/2025.08.20.669565 Yongjie Liu 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Peiwen Xiong 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Songyang Li 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xinjia Liu 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China 2 The School of Pediatrics, Hengyang Medical School, University of South China, (Hunan Children’s Hospital) , Changsha 410007, China ; Find this author on Google Scholar Find this author on PubMed Search for this author on this site Tao Liu 3 Department of Clinical Hematology, College of Pharmacy and Laboratory Medicine Science, Army Medical University , Chongqing 400038, China 4 State Key Laboratory of Trauma and Chemical Poisoning, Institute of Combined Injury, College of Preventive Medicine, Army Medical University , Chongqing, 400038, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Qinglan Yang 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Shuting Wu 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hongyan Peng 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yana Li 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lingling Zhang 5 Institute of Clinical Pharmacology, Anhui Medical University; Key Laboratory of Anti-inflammatory and Immune Medicine, Ministry of Education , Hefei 230032, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yafei Deng 1 Pediatrics Research Institute of Hunan Province, the Affiliated Children’s Hospital of Xiangya School of Medicine, Central South University (Hunan children’s hospital) , Changsha 410007, China 2 The School of Pediatrics, Hengyang Medical School, University of South China, (Hunan Children’s Hospital) , Changsha 410007, China ; Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: youcai.deng{at}tmmu.edu.cn yafeideng01{at}sina.com yongz59{at}icloud.com wangjunping{at}tmmu.edu.cn Yong Zhu 6 College of Basic Medicine, Chongqing Medical University , Chongqing 400016, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: youcai.deng{at}tmmu.edu.cn yafeideng01{at}sina.com yongz59{at}icloud.com wangjunping{at}tmmu.edu.cn Junping Wang 4 State Key Laboratory of Trauma and Chemical Poisoning, Institute of Combined Injury, College of Preventive Medicine, Army Medical University , Chongqing, 400038, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: youcai.deng{at}tmmu.edu.cn yafeideng01{at}sina.com yongz59{at}icloud.com wangjunping{at}tmmu.edu.cn Youcai Deng 3 Department of Clinical Hematology, College of Pharmacy and Laboratory Medicine Science, Army Medical University , Chongqing 400038, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: youcai.deng{at}tmmu.edu.cn yafeideng01{at}sina.com yongz59{at}icloud.com wangjunping{at}tmmu.edu.cn Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstracts Alternative polyadenylation (APA) is a critical posttranscriptional mechanism that generates transcriptomic diversity through the production of mRNA isoforms with distinct 3’ UTRs or coding sequences. Current APA analysis based on single-cell RNA sequencing (scRNA-seq) for establishing cell type-specific APA landscapes primarily rely on Read2 data, which lacks precise cleavage site (CS) information. This limitation restricts their ability to achieve precise de novo mapping of polyadenylation sites (PASs). Here, we present s ingle- c ell P oly A denylation ISO form quantification (scPAISO), a computational pipeline designed for de novo identification of PAS and quantification of PAS isoforms in scRNA-seq data, by leveraging the often-discarded Read1 from 3’ tag-based scRNA-seq protocols. Unlike existing tools, scPAISO directly captures mRNA 3’ end cleavage sites, enabling superior performance in motif enrichment (stronger AAUAAA signal) and peak precision (sharper PAS peaks). Moreover, the smaller peak widths enhance the spatial resolution, enabling more accurate detection of closely spaced PASs in the genome. By integrating Read1 and Read2 data, scPAISO achieves isoform-level quantification with an assignment accuracy exceeding 95%. We demonstrate the robustness of scPAISO in identifying PASs and quantifying APA events across diverse biological contexts, including hematopoiesis, systemic sclerosis (SSc), and mouse tissues. We identified stage-specific 3’ UTR lengthening in hematopoietic progenitors, global 3’ UTR remodeling in SSc and tissue-specific polyadenylation (PA) preference along with RNA-binding proteins in mice. Overall, scPAISO represents a significant advancement in the analysis of APA at single-cell resolution and provides a powerful tool for exploring the regulatory landscape of APA, offering new insights into transcriptome complexity and gene regulation in both health and disease. Introduction The lengths of mRNAs and their 3’ untranslated regions (UTRs) are determined by pre-mRNA cleavage and polyadenylation, processes initiated when polyadenylation site (PAS)-regulating proteins recognize the PAS signal 1 . Most mammalian genes contain multiple PASs, enabling a single gene to produce diverse mRNA isoforms through alternative polyadenylation (APA), thereby significantly contributing to transcriptomic diversity and complexity 2 . In humans, approximately half of all genes utilize different PASs—either proximal or distal PASs within the same 3’ UTR—to generate mRNA isoforms that encode the same protein 3 . Additionally, some genes employ intronic or exonic polyadenylation (iPA or ePA) signals, producing mRNA isoforms that yield distinct protein isoforms 3 . The resulting diverse 3’ ends affect various stages of the RNA life cycle. For instance, variations in 3’ UTR length can alter the availability of binding sites for microRNAs and RNA-binding proteins, thereby regulating mRNA stability, localization, and translation 4 , 5 . Truncated coding or noncoding transcripts may arise through APA at intronic or exonic sites 6 , 7 . APA is tightly regulated, with tissue-specific, cell type-specific, or state-specific patterns 8 – 11 . APA occurs within diverse biological contexts, including cell differentiation 12 , 13 , cell proliferation 14 , embryonic development 8 , and immune responses 15 , 16 , and is dysregulated in diseases 11 , 17 . Existing techniques for direct APA measurement, such as 3’-seq 18 , 3P-seq 19 , PAS-seq 20 , and poly(A)-seq 21 , rely on bulk RNA isolation, yielding only averaged PAS profiles across heterogeneous cell populations. In contrast, single-cell RNA sequencing (scRNA-seq) protocols, such as the 10X Chromium system 22 , capture mRNA 3’ ends via poly(A) priming, thus enabling APA analysis at single-cell resolution. Existing methods (e.g., scAPA 23 , scAPAtrap 24 , Sierra 25 , scDaPars 26 , movAPA 27 , and SCAPTURE 28 ) analyze only Read2, which rarely spans mRNA 3’ end cleavage sites (CSs). Therefore, these methods serve as predictive tools rather than enabling precise PAS mapping. Although Read1 contains CS information, it is typically discarded after barcode/UMI extraction. Here we present s ingle c ell P oly A denylation ISO form quantification ( scPAISO ), a pipeline that leverages both Read1 and Read2 from 3’ tag-based scRNA-seq for de novo mapping and quantification of PASs. scPAISO precisely identifies mRNA 3’ ends from Read1, enabling both known and novel PAS detection (half of detected PASs span <20 bp; 95% <70 bp). By anchoring Read2 to PAS peaks, scPAISO achieves isoform-level quantification, marking a major advancement in single-cell APA analysis. Results Overview of scPAISO In the 10X Chromium 3’ single-cell RNA-seq libraries, the fragment sizes ranged from 300–600 bp. Under PE150 sequencing, Read2 usually contains a 150 bp cDNA sequence, whereas Read1 comprises a 16 bp barcode, 10/12 bp UMI, 30 bp polyT capturer sequence and ∼90 bp cDNA 22 ( Fig. 1a , top ). Typically, only Read2 is subjected to transcriptome mapping for gene-level quantification, whereas Read1—despite containing CS information—is discarded after barcode/UMI extraction. To assess the utility of Read1 for PAS detection, we analyzed the GSE196676 dataset 29 , and mapped both Read1 and Read2 to the genome. For example, in the genomic region chr1:2,401,846-2,405,839, RER1 is located on the sense strand and encompasses four PASs (red), whereas PEX10 is located on the antisense strand and has two PASs (blue) located in the 3’ UTR, and all of those PASs are supported by the polyA_DB database 30 ( Fig. 1a , bottom ). Additionally, the 3’ end of Read2 is broadly distributed, whereas the 5’ end of Read1 (CS position) presents sharper peaks ( Fig. 1a , bottom ). Moreover, two closely spaced proximal PASs of RER1 are clearly distinguished by the Read1 signal, whereas only a single peak is observed in the Read2 signal ( Fig. 1a , bottom ). Compared with the Read2 3’ end, the transcriptome-wide Read1 5’ end is markedly enriched at transcription end sites (TESs), confirming Read1’s superiority for de novo PAS mapping ( Fig. 1b ). Download figure Open in new tab Fig. 1: Overview of the single cell polyadenylation isoform quantification (scPAISO) pipeline for de novo identification and quantification of polyadenylation sites (PASs). ( a ) Schematic representation of the 10X Chromium 3’ single-cell RNA-seq library structure. Read1 contains a barcode, a UMI, a polyT capture sequence, and cDNA, and Read2 consists of a 150 bp cDNA sequence. The sharp peak at the end of Read1 indicates the precise cleavage site (CS) of the RER1 and PEX10 mRNA 3’ ends, enabling accurate PAS identification. Genes on the sense strand and their reads are marked in red, while those on the antisense strand are marked in blue. ( b ) Comparison of Read1 and Read2 enrichment at gene bodies and transcription end sites (TESs). ( c ) Workflow of the scPAISO pipeline, including Read1 alignment, PAS identification, model fitting, and PA isoform quantification. ( d ) Classification of polyadenylation (PA) isoforms into three main categories: exonic polyadenylation (ePA), intronic polyadenylation (iPA), and terminal polyadenylation (tPA). ( e ) Application of scPAISO for differential expression analysis and relative APA usage at single-cell resolution. The scPAISO pipeline comprises the following six steps to systematically identify PASs and quantify PAS usage: (1) STAR-based genome alignment of Read1; (2) extraction of the Read1 5’ end position to generate cleavage site profiles; (3) candidate PAS peak calling using MACS3 based on the above profile 31 ; (4) filtering of internal priming artifacts (6-mers ‘AAAAAA’ within-5/+20 bp of PAS) 8 ; (5) extraction of mapped Read1–Read2 pairs to construct a positive control dataset and fit the distribution of the distance between matched Read2 and the corresponding PAS; and (6) assign all Read2 to PASs based on the distance model and PA isoform counting ( Fig. 1c and Methods ). Based on the position of PASs in transcripts, PA isoforms are divided into three categories: prematurely exonic polyadenylation (ePA) isoforms, prematurely intronic polyadenylation (iPA) isoforms and 3’ terminal polyadenylation (tPA) isoforms ( Fig. 1d ). The resulting single-cell PA isoform quantitative matrix was used for differential expression analysis of PA isoforms and APA analysis ( Fig. 1e ). A single gene may generate multiple PA isoforms, and while differential expression may not be apparent at the gene level, PA isoform-level analyses may reveal significant differences. Consequently, single-cell PA isoform-level quantification provides a more detailed perspective for analyzing differences in cellular expression profiles. scPAISO enables de novo identification of polyadenylation sites with high resolution in hematopoiesis To validate scPAISO, we analyzed 10X Chromium scRNA-seq data from 48,500 bone marrow cells (GSE196676), including approximately 27,000 cells from healthy donors and ∼21,500 cells from patients with immune thrombocytopenia (ITP) 29 . Via graph-based Leiden clustering, and CellTypist annotation 32 , 22 cell populations, such as hematopoietic stem cells (HSC), multipotent progenitors (MPPs), dendritic cells (DCs), and six types of B cell populations, were identified ( Fig. 2a and Supplementary Fig. 1a ). scPAISO identified 30,626 sharp PASs (95% < 69 bp), of which 64.6% overlapped with entries in the public polyA_DB database 30 , and 68.5% were located within 500 bp of annotated PASs. For example, we identified three distinct tPAs within the 1,047 bp 3ʹ UTR of TCF19 , a pair of antisense overlapping tPAs in the extended 3’ UTR of TESK1 and the 3’ UTR of CD72 , and two distinct tPAs within the 278 bp 3ʹ UTR of SNRPG . Despite the short 3’ UTR of SNRPG , Read1 signals allowed the identification of two closely spaced yet clearly distinguishable PASs, whereas Read2 captured only a single peak ( Fig. 2b ). The average 5’ end signal profiles of Read1 and 3’ end signal profiles of Read2 at the PASs further demonstrated that Read1, but not Read2, can be used for accurate detection of PASs ( Fig. 2c ). Among all the genes with predicted PASs, 56% contained alternative PASs ( Fig. 2d ). The majority of PASs were distributed in the 3’ UTR or last exon, with 374 (1.22%) PAS located in extended 3’ UTRs. Additionally, approximately 25.05% of the PASs were in the introns ( Fig. 2e ). Further enrichment analysis revealed a corresponding AAUAAA signal 50 bp upstream of the PAS 27 ( Fig. 2f-g and Supplementary Fig. 1b ). Most annotated PASa contain the AATAAA motif or its single-nucleotide variants, except for PASs located in other exons ( Supplementary Fig. 1b ). Download figure Open in new tab Fig. 2: scPAISO identifies and quantifies PASs in bone marrow hematopoiesis. ( a ) UMAP visualization of scRNA-seq data from bone marrow cells. ( b ) Examples of identified PASs in the 3’ UTR of TCF19 , antisense overlapping PASs in TESK1 and CD72 , and closely spaced PASs in the 3’ UTR of SNRPG . The IGV track was set to ‘autoscale’ for the TCF19 and TESK1 loci, which autoscaled selected tracks to their groupwise maximum. The IGV track maximum was set to 10 for the SNRPG locus. ( c ) Average Read1 and Read2 signal profiles at identified PAS. ( d ) Number of genes with alternative PAS usage. ( e ) Genomic distribution of identified PASs. ( f ) Proportional distribution of PASs containing the AATAAA motif or its 1-nt variants identified by different methods. ( g ) Nucleotide composition surrounding PASs is identified by different methods. ( h ) Peak width distribution of PASs identified by different methods. ( i ) Accuracy of PAS prediction using the gamma distribution model. ( j ) UMAP visualization of cells based on single-cell PA isoform expression, showing high consistency with gene-level clustering. ( k ) Correlations between gene expression matrices from Cell Ranger and scPAISO. Abbreviations: HSC, hematopoietic stem cell; MPP, multipotent progenitor; MEMP, myeloid-erythroid progenitor cell; CycMEMP, cycling MEMP; CMP, common myeloid progenitor; GMP, granulocyte-monocyte progenitor; CLP, lymphoid progenitor cell; CDP, DC progenitor; DC, dendritic cells; PDC, plasmacytoid dendritic cells; ERY, erythroid cell, PreProB, preprogenitor B cell, ProB, progenitor B cell, LateProB, late progenitor B cell. We next compared the reliability of PAS identification between scPAISO and three Read2-based methods (SCAPTURE 28 , scAPAtrap 24 , and Sierra 25 ). PASs identified by scPAISO (30,626 PASs) represented the greatest proportion of the AATAAA motif, followed by those identified by SCAPTURE (36,205 PASs) ( Fig. 2f ). In contrast, methods such as Sierra (127,842 PASs) and scAPAtrap (193,028 PASs) detected many more PASs, including a higher proportion of noncanonical variants, suggesting a trade-off between sensitivity and motif precision. The nucleotide composition of sequences surrounding scPAISO-identified PASs was also more consistent with that reported in previous studies 33 ( Fig. 2g ). To assess the resolution of PAS identification, we compared the distribution of peak widths across different methods. Compared with the other methods, scPAISO yielded the smallest peak widths, indicating higher precision in terms of PAS detection ( Fig. 2h ) . Narrower peaks facilitate the detection of closely spaced PASs, such as the PASs of RER1 ( Fig. 1a ) and SNRPG ( Fig. 2b ). We further compared the distance distributions between the PASs and their adjacent counterparts across different methods and reported that scPAISO could more effectively distinguish adjacent PASs ( Supplementary Fig. 1c ). We then fitted three different distributions to model the distances from the 3’ ends of Read2 to the PASs in the positive control dataset and used these fitting models to predict the most likely PASs for all Read2 sequences. Among these distributions, the gamma distribution had the best fitting effect, with an overall accuracy of 94.9% ( Fig. 2i and Supplementary Fig. 1d-e ). Via the use of this model, we predicted PA isoform expression in each single cell. In some previous studies, Read2 sequences were simply assigned to its closest PAS 8 , 34 , whereas in our positive control dataset, approximately 11% of Read2 sequences contained a PAS that was not the closest ( Supplementary Fig. 1f ). Overall, PA isoforms located in the 3’ UTR tended to have higher expression levels, whereas those located in introns typically presented lower expression levels ( Supplementary Fig. 1g ). The proportion of PA isoforms located in the 3’ UTR and last exon was greater than those of the other regions ( Supplementary Fig. 1h ). Dimensionality reduction and clustering analysis based on single-cell PA isoform expression levels revealed high consistency between gene and PA isoform-level UMAP visualization structures ( Fig. 2j ). Further analysis revealed a strong correlation between Cell Ranger’s gene expression matrix and the merged gene expression matrix of the PA isoforms, indicating that the results of the two methods were consistent ( Fig. 2k ). These results demonstrate that scPAISO not only accurately identifies PASs but also quantitatively measures PA isoform levels at the single-cell level. scPAISO enables the analysis of single-cell PA isoform dynamics during hematopoiesis and ITP-associated APA reprogramming Compared with conventional mRNA quantification, single-cell PA isoform profiling offers a higher-resolution lens for dissecting cellular subpopulations. We used the FindMarkers function from the Seurat package 35 to identify differentially expressed genes or PA isoforms (adjusted p-value 0.3) during hematopoiesis. For some genes, although no significant changes were observed at the gene level, PA isoform alterations were detected during hematopoiesis ( Fig. 3a ). During the differentiation of HSCs into MPPs (HSC→MPP), 354 PA isoforms underwent specific changes, most of which were upregulated. For these ISO-only genes, a consistent trend (upregulated or downregulated) was observed between gene-level changes and PA isoform-level changes; however, the magnitude of changes at the gene level was relatively small ( Fig. 3b ). During the differentiation of progenitor B cells (ProB) into late progenitor B cells (LateProB) (ProB→LateProB), 2,145 PA isoforms significantly changed, most of which were downregulated. GO enrichment analysis revealed that ISO-only PA isoforms involved in the HSC→MPP transition were associated with histone modification, stem cell population maintenance, positive regulation of myeloid cell differentiation, and other processes ( Fig. 3c left ). ISO-only PA isoforms in ProB.G1→LateProB transitions were associated with DNA-templated transcription initiation, RNA methylation, TOR signaling, telomere maintenance, the metaphase/anaphase transition of mitotic cell cycle, and B-cell activation involved in immune responses ( Fig. 3c right ). For example, the expression of histone acetyltransferase 1 ( HAT1 ), a gene that enhances proliferation by coordinating histone production, acetylation, and glucose metabolism 36 , did not significantly differ between HSCs and MPPs. However, one of its tPA isoforms was significantly upregulated in MPPs ( Fig. 3d top ). T-cell acute lymphocytic leukemia 1 ( TAL1 ), which regulates immature human hematopoietic cell self-renewal 37 , also showed no gene-level differential expression between HSCs and MPPs, but one of its iPAs was significantly downregulated in MPPs ( Supplementary Fig. 2a ). TEN1 , a gene essential for telomeric C-strand synthesis 38 exhibited no differential gene expression, but its two nearby tPAs (∼120 bp apart) were highly expressed in proB cells, suggesting a potential role in telomere maintenance ( Fig. 3d bottom ). WDR33 , which encodes a key component of the mRNA polyadenylation machinery 39 , also increased the expression of two tPAs in proB cells, suggesting its involvement in 3’ UTR length regulation during B cell development ( Supplementary Fig. 2b ). Download figure Open in new tab Fig. 3: PA isoform-specific differential expression. ( a ) Number of genes or PA isoforms that are differentially expressed during hematopoiesis. ( b ) Heatmap showing specific changes in PA isoform expression during HSC-to-MPP differentiation and ProB-to-LateProB transition. ISOonly : differential expressions were detected only at the PA isoform level. RNAonly : differential expressions were detected only at the gene level. Both : differential expressions were detected at both the gene and PA isoform levels. ( c ) Gene Ontology (GO) enrichment analysis of specific differentially expressed PA isoforms during HSC differentiation (left) and B cell development (right). ( d ) HAT1 (above) and TEN1 (bottom) showing PA isoform-specific changes without corresponding gene-level expression differences. Read1 track: 5’ ends profile of Read1 from all cells. Read2 track: 3’ ends profile of Read2 from all cells. Cell type track: a predicted 3’ ends profile of Read1 from given cells. ( e ) Number of differentially expressed genes or PA isoforms in ITP. ( f ) Heatmap showing specific changes in PA isoform expression in MPPs (top) and EarlyERYs (bottom). ( g-h ) GO enrichment of upregulated or downregulated PA isoforms in MPPs ( g ) and EarlyERYs ( h ) between normal and ITP cells. ( i ) The expression of CBLB did not significantly differ at the total RNA level in EarlyERYs, but its short isoform was highly expressed in ITP, while the long isoform was expressed at reduced levels. Even broader PA isoform-specific expression changes were observed in each subset between healthy donor and ITP samples ( Fig. 3e ). There were 646 differentially expressed PA isoforms in MPPs and 370 differentially expressed PA isoforms in early erythrocytes (EarlyERYs) ( Fig. 3f ). Upregulated PA isoforms in MPPs from patients with ITP were associated with processes such as mitochondrial ATP synthesis coupled electron transport, erythrocyte differentiation, myeloid cell differentiation, and oxidative phosphorylation, whereas downregulated PA isoforms in MPPs from patients with ITP were associated with processes such as innate immune response-regulating signaling pathway, interferon-beta production and regulation of hematopoietic progenitor cell differentiation ( Fig. 3g ). For example, BRD1 is required for the transcriptional activation of erythroid developmental regulator genes 40 , whose iPA is upregulated in MPPs from patients with ITP ( Supplementary Fig. 2c ). In addition, upregulated PA isoforms in EarlyERYs from patients with ITP were linked to translational elongation and rescue of stalled ribosome, which are closely linked to the hallmark biological processes of early erythropoiesis—ribosome biogenesis and hemoglobin accumulation 41 ; whereas downregulated PA isoforms in EarlyERYs from patients with ITP were associated with hematopoietic progenitor cell differentiation, T cell proliferation, negative regulation of T cell activation, and regulation of cytokine-mediated signaling pathways ( Fig. 3h ). Cbl proto-oncogene B (CBLB) is associated with the biological process of T cell proliferation and it acts as a critical negative regulator of JAK2 signaling in erythroid cells by cooperating with LNK to promote JAK2 degradation and restrain cytokine-induced proliferation 42 ; however, its role in the marked increase of EarlyERYs observed in patients with ITP remains unclear, and no significant difference in CBLB expression was detected at the total RNA level in EarlyERYs. Through differential expression analysis at PA isoform level, we determined that the short isoform of CBLB was upregulated in ITP, whereas the long isoform downregulated in ITP ( Fig. 3i ). This suggests that the switch in PA isoforms may increase the proliferation of EarlyERY by reducing the levels of full-length CBLB protein. Eukaryotic translation initiation factor 4E type 2 (EIF4E2) is involved in the regulation of translation 43 , and the ePA of EIF4E2 is upregulated in EarlyERYs in ITP ( Supplementary Fig. 2d ). SOS1 is a guanine nucleotide exchange factor for Ras proteins and plays a crucial role in the immune response 44 , and its tPA level was significantly was downregulated in EarlyERYs from patients with ITP ( Supplementary Fig. 2e ). In summary, our analysis demonstrated that scPAISO achieves highly accurate quantification at the PA isoform level. Compared with conventional gene-level quantification, this method provides superior resolution in expression profiling, enabling the detection of PA isoform-specific differential expression patterns. This enhanced granularity will significantly facilitate more precise functional genomic studies. Nucleotide-resolution tracking of dynamic PAS selection during hematopoietic differentiation To characterize global patterns of 3’ UTR dynamics during hematopoietic differentiation, we analyzed the above-mentioned bone marrow scRNA-seq data (GSE196676) using PASTA, a computational framework for single-cell PAS heterogeneity analysis 34 . First, we processed the PAS count matrices using Dirichlet multinomial regression to quantify the residual variability of the PA isoforms, which characterized the heterogeneity in the relative use of PASs at single-cell resolution. Dimension reduction was performed directly on the polyA residuals, identifying seven distinct polyA clusters (resolution=0.4). We observed less cell type-specific specificity in polyA clusters ( Fig. 4a-b ). Download figure Open in new tab Fig. 4: Dynamic changes in 3’ UTR length during hematopoiesis. ( a ) UMAP visualization of PA clusters (left) and cell annotation (right) based on PAS usage residuals. ( b ) Matching relationships of cell types between PA usage residual-based clusters and gene-based cell annotation. ( c ) APA rank scores for relative PAS usage across different cell types. Viewed by PAS usage based UMAP (left) and gene expression based UMAP (right). ( d ) CPM (counts per million)-normalized PA isoform expression and APA rank scores of MYH9 (left) and PRKCD (right). ( e ) APA rank score trends during DC differentiation (top). Gene numbers with elongated or shortened 3’ UTR lengths (bottom). ( f-g ) CPM normalized PA isoform expression and PAS preference of LEF1 ( f ) and TOPBP1 ( g ) during DC differentiation. To illustrate changes in the PAS usage ratios, the track in the IGV plot is set to “autoscale”. The utilization ratio of the proximal PAS was explicitly labeled in the diagram. The distance between two adjacent PASs is labeled. We next developed a rank score (ranging from 0–1) to quantify preferential PAS selection for each gene and evaluate the PAS usage bias across hematopoietic lineages. A rank score of 0 indicates exclusive proximal PAS, while a score of 1 indicates exclusive distal PAS. C2 and C3 clusters which had the highest rank scores, included common myeloid progenitors (CMPs), granulocyte–monocyte progenitors (GMPs), DC progenitor 1 (DCP1), EarlyERYs and middle erythroid cells (MiddleERYs), suggesting preferential distal PAS usage ( Fig. 4b-c ). The C7 cluster (lowest rank score) included ProB and LateProB cells, reflecting proximal PAS bias ( Fig. 4b-c ). Next, we systematically identified genes with PAS relative usage dynamics during hematopoiesis via the PASTA package’s FindDifferentialPolyA function (Bonferroni-corrected p 0.05). A total of 6,121 PA isoforms corresponding to 2,001 genes with stage-specific PAS switching were identified. To investigate 3’ UTR length dynamics during hematopoiesis, we aggregated single-cell data by cell type to generate a pseudobulk polyadenylation (PA) quantification matrix. Using this matrix, we computed gene-specific rank scores reflecting PAS usage preference across cell types and performed zscore normalization to standardize the data. We subsequently applied k-means clustering (k=10) to identify distinct patterns of the 3’ UTR length, and we revealed dynamic 3’ UTR length changes throughout hematopoietic differentiation, with stage-specific selection of PASs ( Supplementary Fig. 3a ). Several key regulatory genes exhibited dynamic 3’ UTR lengthening/shortening patterns during hematopoietic differentiation. For example: Myosin heavy chain 9 ( MYH9 ), which critically regulates HSPC survival 45 , and B cell activation and proliferation 46 , exhibited longer 3’ UTRs (preferring the distal PAS) in primitive progenitors (HSCs/MPPs) but shorter 3’ UTRs (preferring the proximal PAS) in committed lineages (e.g., plasmacytoid dendritic cells (PDCs)/B cells) ( Fig. 4d left and Supplementary Fig. 3b ). Notably, the distal PAS was located only 256 bp downstream of the proximal PAS, and this short region contains multiple miRNA binding sites 47 ( Supplementary Fig. 3b ). This not only demonstrates the high resolution of scPAISO in detecting closely spaced PASs but also highlights its ability to identify potential miRNA target 48 and post-transcriptional regulatory mechanisms mediated by APA. Protein kinase C delta ( PRKCD ), a key modulator of HSPC proliferation and metabolism 49 , similarly showed a distal PAS (only 157 bp away from the proximal PAS) preference in undifferentiated HSPCs ( Fig. 4d right and Supplementary Fig. 3c ). PolyA residual based clustering revealed a continuous APA usage change from HSCs (C1), MPPs (C1), CMPs (C3), GMPs (C3), DCP1 (C3), and DCs (C4) to PDCs (C4). To investigate PAS usage dynamics during HSC differentiation into PDCs, we quantified rank scores across developmental stages. The rank scores gradually increased from HSC stage and peaked at CMP stage but then decreased later in differentiation ( Fig. 4e , top ). We then clustered genes showing significant PAS usage changes (Bonferroni-corrected p 0.05) and identified the most significant changes in PAS usage occurring during the transition from MPP to CMP ( Fig. 4e , bottom ). For example, LEF1, a transcription factor for myeloid progenitor proliferation and differentiation 50 , showed greater distal PAS preference (high rank score) in CMPs/GMPs ( Fig. 4f ). To illustrate the changes in PAS usage ratios, the track scale in the IGV was set to ‘autoscale’. The utilization ratio of the proximal PAS was explicitly labeled. The proximal PAS of TEL1 accounted for 29% and 16% in HSCs and MPPs, respectively, but accounted for only 2% and 11% in CMPs and GMPs. Similarly, TOPBP1, which encodes a key DNA topoisomerase for DNA replication/repair processes 51 , also exhibited a preference for a distal PAS (only 352 bp away from the proximal PAS) in CMPs/GMPs ( Fig. 4g ). PAS usage preferences in SSc SSc is a progressive autoimmune disorder characterized by pathological collagen deposition in cutaneous and visceral tissues. Using scPAISO, we analyzed scRNA-seq data from the skin tissue of 22 patients with SSc and 18 healthy donors 52 . After stringent quality filtering (retaining 95,954 high-quality cells), we categorized the transcriptome into 12 major cell types ( Fig. 5a and Supplementary Fig. 4a ). scPAISO identified 33,683 sharply PAS peaks, with 95% localized within 72-bp regions. For example, RCAN3 exhibited three specific tPAs in its 3’ UTR ( Supplementary Fig. 4b ). Nucleotide composition near the PAS aligned with the canonical motif ( Supplementary Fig. 4c ). When assigning Read2 to PAS, a Read2 assignment accuracy of 96.78% was achieved using Weibull distribution modeling ( Supplementary Fig. 4d ). UMAP visualization of cell-type signatures confirmed strong concordance between gene and PA-isoform-level clustering ( Fig. 5a ). Download figure Open in new tab Fig. 5: PA isoform preference in systemic sclerosis (SSc). ( a ) UMAP visualization of gene expression (left) and PA isoform (right) data from patients with SSc and healthy controls. ( b ) Number of differentially expressed genes or PA isoforms in SSc. ( c ) GO enrichment analysis of specific differentially expressed PA isoforms in Fibroblast.1 (left) and T cells (right). ( d ) TNFRSF1A (top) and IER3IP1 (bottom) showing PA isoform-specific changes without corresponding gene-level expression differences. ( e ) Gene numbers with dynamic changes in 3’ UTR length between SSc and controls. ( f ) GO enrichment analysis of genes with elongated or shortened 3’ UTRs in T cells. ( g ) PAS preference changes in CD28, FOXO3 and IL16 . The distance between two adjacent PASs is labeled. Disease-associated PAS usage analysis revealed widespread differential expression of PA isoforms without corresponding changes in total gene expression across most cell populations ( Fig. 5b ), highlighting posttranscriptional dysregulation. In Fibroblast.1, differentially expressed PA isoform-specific genes were implicated in macroautophagy, the ERBB signaling pathway, and regulation of TORC1 signaling ( Fig. 5c ). Notably, both IER3 interacting protein 1 ( IER3IP1 ) and tumor necrosis factor receptor superfamily member 1A ( TNFRSF1A ), which are involved in the regulation of fibroblast apoptosis 53 , 54 , were differentially expressed PA isoform-specific genes ( Fig. 5d ). In T cells, differentially expressed PA isoform-specific genes were linked to regulation of T cell activation and differentiation, leukocyte cell–cell adhesion, and response to type I interferon ( Fig. 5c ). To further investigate 3’ UTR length remodeling in SSc, we quantified rank scores and compared healthy donors and patients with SSc across cell populations. Strikingly, global 3’ UTR lengthening was observed in disease-associated cell types, including Fibroblast.1, T cell, and myeloid lineages ( Fig. 5e ). Functional implication analysis of 3’ UTR remodeling in T cells revealed that genes with elongated 3’ UTRs were enriched for interleukin-4 production and positive regulation of alpha-beta T cell proliferation, while genes with shortened 3’ UTRs were strongly associated with positive regulation of interleukin-2 and interleukin-12 production, and regulation of inflammatory responses ( Fig. 5f ). Key regulatory genes in T cells exhibited differential PAS usage. CD28 (T cell costimulation and activation) and FOXO3 (critical for T cell homeostasis and survival) 55 , 56 were associated with a distal PAS preference in SSc. In contrast, IL16 , which is involved in immune response, preferentially used the proximal PAS in SSc 57 ( Fig. 5g ). PAS usage landscape across mouse tissues Although PA dynamics during mouse embryogenesis have been explored 8 , a systematic characterization of PAS usage—particularly at single-cell resolution—remains incomplete for both the embryonic and adult stages, owing to limitations in precise PAS mapping. To address this gap, we implemented scPAISO on a comprehensive scRNA-seq dataset containing cells from brain, liver, muscle, and skin tissues and hematopoietic lineages (isolated from bone marrow, peripheral blood, and the spleen) 58 . After quality filtering (retaining 265,947 cells) and tissue-specific dimensionality reduction and annotation ( Supplementary Fig. 5a ), we integrated the data and reanalyzed the cells for unified visualization ( Fig. 6a ). Next, we applied scPAISO to this dataset and identified a total of 51,926 PASs. The nucleotide composition surrounding the 3’ UTR PASs was consistent with those in previous reports ( Supplementary Fig. 5b ). When Read2 data were assigned to PAS, the Weibull distribution yielded the best fit, achieving an overall accuracy of 95.91% ( Supplementary Fig. 5c ). Notably, we observed distinct tissue-specific PAS usage patterns: immune cells displayed systematically higher rank scores, suggesting a bias toward distal PAS selection, whereas parenchymal cells—particularly those from the liver, brain, and muscle—showed lower rank scores, indicating a preference of the proximal PAS ( Fig. 6a ). Download figure Open in new tab Fig. 6: Tissue and cell-specific PAS usage preferences in mouse tissues. ( a ) UMAP visualization of cell annotation (left) and APA rank scores (right) from the scRNA-seq data of multiple mouse tissues. ( b ) Left: heatmap of APA rank score patterns of genes with significant PAS usage changes. Right: heatmap of gene expression patterns of mRNA 3’ end processing factors. Abbreviations: Brain: EN, excitatory neurons; IN, inhibitory neurons; NPC, neural progenitor cells; Oligo, oligodendrocytes. Skeletal muscle: FAP, fibro-adipogenic progenitor cells; FAST, fast-twitch muscle fibers; MTJ, myotendinous junction; NMJ, neuromuscular junction; MuSCs, muscle stem cells. Liver: Ito, Hepatic stellate cells. Skin: Basal, stratum basale keratinocyte; EC, endothelial cell; HFSCs, hair follicle stem cells; HF, hair follicle; IB, inner bulge; PapiFib, papillary fibroblast; RetiFib, reticular fibroblast; SG, sebaceous gland cells; SMC, smooth muscle cell; Spi, stratum spinosum keratinocyte. Hematopoietic/immune system: HSC, hematopoietic stem cell; MPP, multipotent progenitors; MEP, myeloid-erythroid progenitor; CMP, common myeloid progenitor; GMP, granulocyte-monocyte progenitor; CLP, lymphoid progenitor. To further characterize PAS usage heterogeneity across diverse cell types, we identified 5,220 genes that exhibited significant variation in 3’ UTR PAS selection. Based on the result of the rank score analysis, these genes were clustered into eight modules, and the cells were segregated into six major categories ( Fig. 6b , left ). Clustering analysis revealed that immune lineages (primarily C5 and C6) and dermal parenchymal cells (primarily C4) exhibited coordinated PAS selection profiles. Despite distinct ontogeny, parenchymal cells from striated muscle and hepatocytes (C3) were clustered together with comparable PAS rank score distributions ( Fig. 6b , left ). Striking examples of cell type-specific PAS switching included: Foxo3 , which was highly expressed in C1 and C2 but showed preferential distal PAS usage (higher rank scores) in C4 to C6, implying uncoupled regulation of transcript levels and 3’ UTR isoform selection ( Supplementary Fig. 6a ); Pbx2 used the distal PAS from C1 to C4 but shifted to proximal PAS in C5 and C6 ( Supplementary Fig. 6b ). To explore the transcriptional programs governing PAS selection dynamics across cell types, we profiled the expression of core mRNA 3’ end processing factors 34 and subjected them to unbiased clustering ( Fig. 6b , right ). This analysis resolved eight gene modules (A1–A8), each of which demonstrated pronounced cell-type specificity. For example, Cstf2 (module A1) and Pabpn1 (module A3) were highly expressed in neural cells, Cpeb1 (module A4) was highly expressed in C3 cells, and included liver and muscle tissue, while cleavage and polyadenylation specificity factor 2 ( Cpsf2 ) and serine/arginine-rich splicing factor 3 ( Srsf3 ) 59 (module A8) were highly expressed in immune cells ( Supplementary Fig. 6c ). These findings suggest that cell type-specific PAS preferences are likely shaped by the combined expression dynamics of these regulatory modules. Coupling between RNA-binding protein (RBP) expression and 3’ UTR length The extension of 3’ UTR length may lead to an increase in RBP binding sites. Therefore, it is reasonable to hypothesize that the overall elongation of 3’ UTRs in cells might be coupled with upregulation of the expression levels of corresponding RBPs. Accordingly, we performed k-means clustering on all annotated 1,644 RBP genes 60 (except for core mRNA 3’ end processing factors) based on their expression profiles, resulting in ten distinct modules (R1 to R10). Module R2 was highly expressed in cell clusters C2 and C3 ( Fig. 7a ), and correspondingly, genes in Cluster M2 exhibited longer 3’ UTRs in these cell clusters ( Fig. 6b , left) . Module R3 was highly expressed in cell cluster C2 ( Fig. 7a ), and genes in Cluster M1 displayed longer 3’ UTRs specifically in C2 ( Fig. 6b , left) . Module R10 was predominantly expressed in cell clusters C4–C6 ( Fig. 7a ), and genes in clusters M6 and M7 had longer 3’ UTRs in these cell populations ( Fig. 6b ). For example, the expression of the Exosc family was increased in immune cell types (C5 and C6), whereas the expression levels of the Dnah and Celf gene families were increased in brain tissue-derived cells (C2) ( Fig. 7b ) . We propose that the upregulated expression of specific RBPs may be functionally linked to the elongation of target gene 3’ UTRs, suggesting a potential posttranscriptional regulatory mechanism. RBPs exhibit specific sequence preferences in terms of RNA binding 61 , 62 . To further confirm the coupling between RBP expression and 3’ UTR length, we collected motif information of RBPs from starBase v2.0 47 and analyzed the enrichment of RBP motifs in 3’ UTRs across different cell types. Specifically, we downloaded the top-ranked motifs of different RBPs from various crosslinking-immunoprecipitation (CLIP) experiments in the starBase database, which yielded in 2,179 motif entries corresponding to 319 RBP genes ( Supplementary Fig. 7a ). Furthermore, MOODS-DNA 63 was used for genome-wide motif scanning, and strand-matched motif information within 3’ UTRs was subsequently extracted for downstream analysis. We then extracted the genomic coordinates of the 3’ UTRs of different tPA isoforms, defined them as peaks and quantified the expression levels of these tPA isoforms as peak counts. These data were subsequently converted into a ChromVAR-compatible object 64 . Using the ChromVAR package, we ultimately quantified the enrichment of RBP binding motifs across different cell types. Download figure Open in new tab Fig. 7: Gene expression patterns of RBP genes. ( a ) Heatmap of the gene expression patterns of RBP-related genes. ( b ) Gene expression preferences of Exosc -family, Dnah -family and Celf -family genes. ( c-d ) Gene expression levels and motif enrichment of Rnps1 ( c ) and G3bp1 ( d ) across cell clusters. Among these RBP-motif pairs, 169 showed a positive correlation between gene expression and corresponding motif enrichment (Pearson correlation > 0.5) ( Supplementary Fig. 7b ). For example, RNA binding protein with serine repeat 1 (Rnps1), which is involved in mRNA splicing and RNA quality control, regulates hematopoietic stem cell (HSC) differentiation as well as B and T cell development and function 65 . This gene was highly expressed in cluster C5 (mainly immune cells), while its motif was also more highly enriched in the corresponding group ( Fig. 7c ) . Another example is GTPase activating protein binding protein 1 (G3bp1), which regulates mRNA stability and translation. G3bp1 plays a role in modulating inflammatory responses in macrophages and dendritic cells, and can influence antibody secretion in B cells 66 , 67 . It is highly expressed in clusters C4–C6, with its motif showing greater enrichment in the corresponding cell groups ( Fig. 7d ). These results support a coordinated relationship between 3’ UTR elongation and increased expression of corresponding RBPs across cell types. The positive correlation between RBP expression and motif enrichment further suggests a potential posttranscriptional regulatory mechanism underlying 3’ UTR lengthening. Discussion In this study, we present scPAISO, a novel computational pipeline designed for the de novo identification and quantitative analysis of PASs at single-cell resolution through 3’ tag-based scRNA-seq data. By leveraging Read1—a data source that encompasses vital 3’ cleavage site information yet is conventionally discarded in standard scRNA-seq analyses—scPAISO enables precise PAS localization and quantitation of PA isoforms within individual cells. Previously, scRNA-seq APA pipelines inferred PAS positions from diffuse Read2 peaks, yielding ambiguous coordinates 24 , 25 , 28 . scPAISO directly utilizes Read1 to resolve PASs into sharp, discrete peaks, markedly improving positional precision, especially for intron-crossing signals and closely spaced PASs. In the RER1 locus ( Fig. 1a ), an intron within the 3’ UTR generates a strong Read2 signal at its 5’ splice site that could be misannotated as a PAS. Read1 data confirmed that there is no internal PAS, demonstrating that the flanking reads originate from a single continuous transcript. In addition, scPAISO clearly distinguished two closely spaced proximal PASs undetectable in Read2-based analyses ( Fig. 1a ). Moreover, scPAISO PAS calls align precisely with polyA_DB 30 annotations, underscoring the method’s fidelity and clarity ( Fig. 1a ). Furthermore, scPAISO outperforms prior methods by identifying significantly more de novo PASs and enabling quantitative tracking of PA isoforms neglected by indirect inference tools. In quantifying PA isoforms, scPAISO attained >95% allocation accuracy and yielded PAS peaks that were markedly sharper (95% width <70 bp) than those produced by Read2-based tools 24 , 25 , 28 . This narrow peak profile enables reliable discrimination of adjacent PASs and resolves genes with abbreviated 3′ UTRs. For example, in CD28 3’ UTR ( Fig. 5i ), Read2 showed a dense, ambiguous cluster of signals, whereas Read1 resolved two discrete high-confidence PASs. A similar resolution is achieved for TAL1 ’s iPAS ( Fig. s2a ). Reexamination of these previously discarded Read1 signals markedly enriched canonical CPSF-complex binding motifs 68 : the frequencies of AATAAA and ATTAAA increased sharply around scPAISO-defined PASs compared with those around PASs obtained via other tools 24 , 25 , 28 ( Fig. 2f ). Concomitantly, artifactual runs of adenosines, often misassigned as PASs by the impact of internal primer dimers, were significantly reduced. Applications of scPAISO in hematopoiesis, SSc, and mouse multitissue atlases revealed unannotated PASs 30 in genes such as TAL1 37 , MYH9 45 , PRKCD 49 (hematopoiesis), and SOS1 44 , CD28 55 , and FOXO3 56 (immune response and activation). scPAISO applied to multiple mouse tissues also revealed cell type-and tissue-specific preferences for PAS usage (e.g., immune cells with a distal PAS preference, parenchymal cells in brain and liver with a proximal PAS preference) ( Fig. 6b ). The expanded PASs obliges us to reassess how APA impacts the expression and activity of these pivotal genes. The selective use of these PASs can (i) reposition miRNA and RBP-binding sites or alter epigenetic marks within the affected 3′ UTR (tPA) or (ii) generate distinct protein isoforms (iPA and ePA) 3 . Beyond the direct posttranscriptional effects of APA on gene output, specific APA signatures can serve as orthogonal biomarkers for cell-state classification during the initial clustering step of single-cell workflows. PA isoform analysis reveals expression differences that are not detectable by gene-level analysis alone. These APA-only differences recapitulate both lineage relationships and disease-associated alterations, underscoring the complementary power of single-cell APA profiling. CBLB acts as a critical negative regulator of EarlyERY proliferation 42 . Although no significant difference in the expression of CBLB was detected at the total RNA level in EarlyERYs, its short isoform was highly expressed in ITP, whereas the long isoform showed reduced expression in ITP, suggesting that the switch in PA isoforms may enhance the proliferation of EarlyERYs by reducing the levels of the full-length CBLB protein ( Fig. 3i ). In SSc, high-resolution PAS mapping and precise quantification by scPAISO also revealed that APA dynamics could refine disease subtyping with clinical relevance. Moreover, integrating PA-isoform dynamics into clustering algorithms thus offers an extra dimension to refine cell-type annotation and to chart subtle developmental transitions that would otherwise remain invisible. In conclusion, in scPAISO, discarded Read1 data are repurposed to deliver single-cell PAS atlases with nucleotide resolution, uncovering APA’s roles in developmental regulation (e.g., hematopoiesis), disease mechanisms (e.g., SSc stratification), and therapeutic targeting (e.g., isoform-specific interventions). As single-cell technologies continue to evolve, tools such as scPAISO will be helpful for unraveling the complexity of the transcriptome and its regulation at the single-cell level. Moreover, current studies on 3’ UTR alternative polyadenylation quantitative trait loci (3’aQTLs) are all based on bulk RNA-seq data 69 , which inherently have significant limitations in calculating PAS usage. The development of our method could facilitate research on 3’aQTLs that is based on published scRNA-seq data 70 – 72 . Methods Processing of scRNA-seq data Single-cell RNA-seq raw fastq data were obtained from publicly available datasets, including bone marrow hematopoiesis (GEO, GSE196676) 29 , SSc (GEO, GSE249279) 52 , and mouse tissues (GSA, CRA004660) 58 . The single-cell expression matrix was generated using Cell Ranger ( https://github.com/10XGenomics/cellranger ; GRCh38 for human, GRCm38 for mouse) and processed with the Seurat package (v4.0.1) 35 . In hematopoiesis, low-quality cells with fewer than 2000 genes detected or more than 15% mitochondrial reads were excluded. Genes expressed in fewer than 10 cells were also removed. The remaining cells were normalized using the LogNormalize method with a scale factor of 10,000. The top 3,000 highly variable genes were selected for downstream analysis. Principal component analysis (PCA) was performed on the scaled data, and the top 30 principal components were used as inputs for Harmony (v0.1.0) integration 73 . The top 20 integrated embeddings from Harmony were then used for subsequent clustering and visualization. Cell clusters were identified at resolution = 0.8. Clusters with < 500 cells were removed. CellTypist (v1.6.3) 32 was used for cell annotation with default parameters and was validated manually against known marker genes. In SSc, low-quality cells with 50000 UMIs, > 40% ribosome reads, or > 15% mitochondrial reads were excluded. Genes expressed in < 10 cells were also removed. The top 2,000 highly variable genes were selected for downstream analysis. PCA was performed on the scaled data, and the top 50 principal components were used as inputs for Harmony (v0.1.0) integration. The top 20 integrated embeddings were employed for subsequent clustering and visualization. Cell clusters were identified at resolution = 0.6; clusters < 500 cells were removed for further analysis. Cell annotation was performed based on marker genes from prior studies. In mouse tissues, low-quality cells with 40000 UMIs, 6000 genes, > 40% ribosome reads, or > 15% mitochondrial reads were excluded. Genes expressed in < 5 cells were also removed. Cells were classified into muscle, liver, brain, skin, and hematopoietic groups based on tissue type and sorting method. Integration, clustering, and cell annotation were performed for each group as in the SSc. After annotation, all the cells were merged; the top 6,000 highly variable genes were used for PCA and the top 50 principal components were used for UMAP. scPAISO pipeline Read1 alignment For each library, Read1 was aligned to the reference genome (GRCh38 for human, GRCm38 for mouse) using STAR aligner (version 2.7.10b) with the following parameters: --clip5pNbases 58--outFilterMismatchNmax 999--outFilterMismatchNoverLmax 0.3--outFilterScoreMinOverLread 0--outFilterMatchNminOverLread 0--outFilterMatchNmin 70 . Read2 alignment was performed using the Bam file output by Cell Ranger. Identification of PASs BAM processing: scanBam ( https://github.com/Bioconductor/Rsamtools ) was used to read the BAM files, and only paired reads and uniquely mapped reads were retained. Paired reads with a distance > 1 Mb between their mapped positions on the chromosome were filtered. Paired reads were aligned to genes based on Read2 positions; intron lengths were subtracted from Read2–Read1 distances. Paired reads with a distance > 10,000 bp were filtered out. PAS peak calling: the 5’ ends of Read1 (sense and antisense strands) were extracted separately and exported as BigWig files using the‘export.bw‘function from the rtracklayer package. MACS3 (version 3.0.0) 31 identified candidate PASs with the following parameters: --max-gap 10--min-length 5--extsize 3--shift-1--call-summits--keep-dup all-p 0.00001. Candidate PASs with the 6-mer’AAAAAA’ within 5 bp upstream to 20 bp downstream were filtered to exclude priming artifacts. PASs within 50 bp were further merged into the final PAS. PAS annotation: PASs were classified as 3’ UTRs, introns, or exons based on their genomic location. The 3’ UTR PASs were further classified as terminal PASs (tPAs), whereas those in introns or exons were classified as intronic PASs (iPAs) or exonic PASs (ePAs), respectively. Model fitting and PA isoform quantification For the previously described paired Read1 and Read2 sequences, those where Read1 did not overlap with the PAS were excluded, generating a positive control dataset. The intron-removed distance between Read2 and its corresponding PAS was calculated and used to fit a statistical model (Weibull, Gamma, or Log-normal distribution). The fitted model was applied to all Read2 in the positive control datasets to predict the most likely PAS for each read. The model with the highest global accuracy was subsequently used for the quantification of PA isoforms. A single-cell PA isoform count matrix was then generated, where each cell was represented by the expression levels of different PAS isoforms. This matrix was used for downstream analysis, including differential expression, clustering and PAS usage analysis. Rank score calculation A rank score was calculated for each gene to evaluate the usa of different-length PAS isoforms to quantify PAS usage bias: Where n denotes the number of PAs, 𝑝 i denotes the proportion of expression of the i-th PA relative to the total expression of the gene. The rank score ranges from 0 (exclusive use of the proximal PAS) to 1 (exclusive use of the distal PAS). Cell clustering by relative PAS usage The PA isoform count matrix was processed using PASTA 34 . And CalcPolyAResiduals 34 was used to calculate the residuals of PA isoforms based on the Dirichlet multinomial distribution, which can characterize heterogeneity in the relative use of PASs at single-cell resolution. FindVariableFeatures 35 was used to identify polyA sites that varied across cells, after which dimension reduction was performed on the variable polyA sites. Differential analysis For gene and PA isoform-level differential expression analysis, the Seurat package (version 4.0.5) was used. The FindMarkers 35 function was used to identify differentially expressed genes or PAS isoforms between cell types or conditions (p_val_adj 0.3). Differential relative PAS usage analysis was performed using FindDifferentialPolyA function from PASTA 35 . This function returns the following outputs: estimate , estimated coefficient from the differential APA linear model (indicates the magnitude of change); p.value , p-value from the linear model, p_val_adj , Bonferroni-corrected p-value; percent.1 , the average percent usage (pseudobulk) of the polyA site in the first group of cells; percent.2 , the average percent usage (pseudobulk) of the polyA site in the second group of cells. A threshold of p_val_adj 0.05 were used for further filtering. To compare the overall changes in 3’ UTR length among different conditions, FindDifferentialPolyA was used to identify genes whose relative PAS usage changed. For each gene, we multiplied the estimate by the ordinal number of the PAS and then summed the values. If the result was > 0, it indicated that the overall 3’ UTR length had increased; if < 0, it indicated that the overall 3’ UTR length had decreased. Data availability The single-cell RNA-seq datasets used in this study are available from the Gene Expression Omnibus (GEO) or Genome Sequence Archive (GSA) under the accession numbers GSE196676 (bone marrow hematopoiesis), GSE249279 (SSc), and CRA004660 (mouse tissues). The processed data and analysis results are available upon request. Code availability The scPAISO pipeline, including scripts for data processing, PAS identification, and quantification, is available on GitHub ( https://github.com/yongjieliu/scPAISO ). Detailed documentation and example datasets are provided to facilitate the use of the proposed pipeline. Author contributions Yongjie Liu, Youcai Deng, and Yafei Deng conceived and supervised the project. Yongjie Liu and Peiwen Xiong performed the data analyses. Yongjie Liu, Youcai Deng, Yafei Deng, Junping Wang and Yong Zhu interpreted the data and wrote the manuscript. Songyang Li, Xinjia Liu, Tao Liu, Qinglan Yang, Shuting Wu, Hongyan Peng, Yana Li, and Lingling Zhang reviewed and revised the manuscript. Competing interests The authors declare that they have no competing interests. Acknowledgments This study was supported by funding from the National Natural Science Foundation of China (82273938, 81922068, 82473934, 32100616, 82304511, 82473935); National Key Research and Development Project (2020YFA0113500); Hunan Provincial Natural Science Foundation of China (2023JJ40341, 2023JJ40352); Scientific Research Project of Hunan Provincial Health Commission (A202302077875, B202302078452, China); the Projects of Army Medical University (2022XJS06, 2023XQN20, 2023XQN19, China); the Chongqing Municipal Education Commission Foundation (KJZD-M202412802, China); the Natural Science Foundation of Changsha, Hunan Province (kq2208087, China); and the Key Research and Development Program Project of the State Key Laboratory of Trauma and Chemical Poisoning (2024K004). We gratefully acknowledge the use of publicly available single-cell RNA-seq datasets from the Gene Expression Omnibus (GEO) and the Genome Sequence Archive (GSA). Specifically, data with the accession numbers GSE196676, GSE249279 and CRA004660 were utilized in this study. These datasets provided important resources for method validation and greatly facilitated our research. Writing errors were checked using Grammarly ( http://grammarly.com/ ). References 1. ↵ Liu , H. & Moore , C. L . On the Cutting Edge: Regulation and Therapeutic Potential of the mRNA 3’ End Nuclease . Trends Biochem. Sci . 46 , 772 – 784 ( 2021 ). OpenUrl CrossRef PubMed 2. ↵ Derti , A. et al. A quantitative atlas of polyadenylation in five mammals . Genome Res . 22 , 1173 – 1183 ( 2012 ). OpenUrl Abstract / FREE Full Text 3. ↵ Mitschka , S. & Mayr , C . Context-specific regulation and function of mRNA alternative polyadenylation . Nat. Rev. Mol. Cell Biol . 23 , 779 – 796 ( 2022 ). OpenUrl CrossRef PubMed 4. ↵ Mayr , C . Evolution and Biological Roles of Alternative 3’UTRs . Trends Cell Biol . 26 , 227 – 237 ( 2016 ). OpenUrl CrossRef PubMed 5. ↵ Soles , L. V. et al. A nuclear RNA degradation code is recognized by PAXT for eukaryotic transcriptome surveillance . Mol. Cell 85 , 1575 – 1588 .e9 ( 2025 ). OpenUrl PubMed 6. ↵ Lee , S.-H. et al. Widespread intronic polyadenylation inactivates tumour suppressor genes in leukaemia . Nature 561 , 127 – 131 ( 2018 ). OpenUrl CrossRef PubMed 7. ↵ Singh , I. et al. Widespread intronic polyadenylation diversifies immune cell transcriptomes . Nat. Commun . 9 , 1716 ( 2018 ). 8. ↵ Agarwal , V. , Lopez-Darwin , S. , Kelley , D. R. & Shendure , J . The landscape of alternative polyadenylation in single cells of the developing mouse embryo . Nat. Commun . 12 , 5101 ( 2021 ). 9. Lee , S. et al. Diverse cell-specific patterns of alternative polyadenylation in Drosophila . Nat. Commun . 13 , 5372 ( 2022 ). 10. Chen , H. et al. A distinct class of pan-cancer susceptibility genes revealed by an alternative polyadenylation transcriptome-wide association study . Nat. Commun . 15 , 1729 ( 2024 ). 11. ↵ Gruber , A. J. & Zavolan , M . Alternative cleavage and polyadenylation in health and disease . Nat. Rev. Genet . 20 , 599 – 614 ( 2019 ). OpenUrl CrossRef PubMed 12. ↵ Sommerkamp , P. et al. Differential Alternative Polyadenylation Landscapes Mediate Hematopoietic Stem Cell Activation and Regulate Glutamine Metabolism . Cell Stem Cell 26 , 722 – 738 .e7 ( 2020 ). OpenUrl CrossRef PubMed 13. ↵ Cheng , L. C. et al. Widespread transcript shortening through alternative polyadenylation in secretory cell differentiation . Nat. Commun . 11 , 3182 ( 2020 ). 14. ↵ Sandberg , R. , Neilson , J. R. , Sarma , A. , Sharp , P. A. & Burge , C. B . Proliferating cells express mRNAs with shortened 3’ untranslated regions and fewer microRNA target sites . Science 320 , 1643 – 1647 ( 2008 ). OpenUrl Abstract / FREE Full Text 15. ↵ L, L., et al. Immune-response 3’UTR alternative polyadenylation quantitative trait loci contribute to variation in human complex traits and diseases . Nat. Commun . 14 , ( 2023 ). 16. ↵ Ge , Y. et al. Downregulation of CPSF6 leads to global mRNA 3’ UTR shortening and enhanced antiviral immune responses . PLOS Pathog . 20 , e1012061 ( 2024 ). OpenUrl CrossRef PubMed 17. ↵ Witkowski , M. T. et al. NUDT21 limits CD19 levels through alternative mRNA polyadenylation in B cell acute lymphoblastic leukemia . Nat. Immunol . 23 , 1424 – 1432 ( 2022 ). OpenUrl CrossRef PubMed 18. ↵ Lianoglou , S. , Garg , V. , Yang , J. L. , Leslie , C. S. & Mayr , C . Ubiquitously transcribed genes use alternative polyadenylation to achieve tissue-specific expression . Genes Dev . 27 , 2380 – 2396 ( 2013 ). OpenUrl Abstract / FREE Full Text 19. ↵ Jan , C. H. , Friedman , R. C. , Ruby , J. G. & Bartel , D. P . Formation, regulation and evolution of Caenorhabditis elegans 3’UTRs . Nature 469 , 97 – 101 ( 2011 ). OpenUrl CrossRef PubMed Web of Science 20. ↵ Shepard , P. J. et al. Complex and dynamic landscape of RNA polyadenylation revealed by PAS-Seq . RNA N. Y. N 17 , 761 – 772 ( 2011 ). OpenUrl 21. ↵ Derti , A. et al. A quantitative atlas of polyadenylation in five mammals . Genome Res . 22 , 1173 – 1183 ( 2012 ). OpenUrl Abstract / FREE Full Text 22. ↵ Zheng , G. X. Y. et al. Massively parallel digital transcriptional profiling of single cells . Nat. Commun . 8 , 14049 ( 2017 ). 23. ↵ Shulman , E. D. & Elkon , R . Cell-type-specific analysis of alternative polyadenylation using single-cell transcriptomics data . Nucleic Acids Res . 47 , 10027 – 10039 ( 2019 ). OpenUrl CrossRef PubMed 24. ↵ Wu , X. , Liu , T. , Ye , C. , Ye , W. & Ji , G . scAPAtrap: identification and quantification of alternative polyadenylation sites from single-cell RNA-seq data . Brief. Bioinform . 22 , bbaa273 ( 2021 ). 25. ↵ Patrick , R. et al. Sierra: discovery of differential transcript usage from polyA-captured single-cell RNA-seq data . Genome Biol . 21 , 167 ( 2020 ). 26. ↵ Gao , Y. , Li , L. , Amos , C. I. & Li , W . Analysis of alternative polyadenylation from single-cell RNA-seq using scDaPars reveals cell subpopulations invisible to gene expression . Genome Res. gr . 271346 . 120 ( 2021 ). 27. ↵ Ye , W. et al. movAPA: modeling and visualization of dynamics of alternative polyadenylation across biological samples . Bioinformatics 37 , 2470 – 2472 ( 2021 ). OpenUrl CrossRef PubMed 28. ↵ Li , G.-W. et al. SCAPTURE: a deep learning-embedded pipeline that captures polyadenylation information from 3′ tag-based RNA-seq of single cells . Genome Biol . 22 , 221 ( 2021 ). 29. ↵ Liu , Y. et al. Deciphering transcriptome alterations in bone marrow hematopoiesis at single-cell resolution in immune thrombocytopenia . Signal Transduct. Target. Ther . 7 , 1 – 18 ( 2022 ). OpenUrl CrossRef PubMed 30. ↵ Wang , R. , Zheng , D. , Yehia , G. & Tian , B . A compendium of conserved cleavage and polyadenylation events in mammalian genes . Genome Res . 28 , 1427 – 1441 ( 2018 ). OpenUrl Abstract / FREE Full Text 31. ↵ Zhang , Y. et al. Model-based analysis of ChIP-Seq (MACS) . Genome Biol . 9 , R137 ( 2008 ). 32. ↵ Domínguez Conde , C. , et al. Cross-tissue immune cell analysis reveals tissue-specific features in humans . Science 376 , eabl5197 ( 2022 ). 33. ↵ Ye , W. , Lian , Q. , Ye , C. & Wu , X . A Survey on Methods for Predicting Polyadenylation Sites from DNA Sequences, Bulk RNA-seq, and Single-cell RNA-seq . Genomics Proteomics Bioinformatics 21 , 67 – 83 ( 2023 ). OpenUrl CrossRef PubMed 34. ↵ Kowalski , M. H. et al. Multiplexed single-cell characterization of alternative polyadenylation regulators . Cell 187 , 4408 – 4425 .e23 ( 2024 ). OpenUrl CrossRef PubMed 35. ↵ Hao , Y. et al. Integrated analysis of multimodal single-cell data . Cell 184 , 3573 – 3587 .e29 ( 2021 ). OpenUrl CrossRef PubMed 36. ↵ Gruber , J. J. et al. HAT1 Coordinates Histone Production and Acetylation via H4 Promoter Binding . Mol. Cell 75 , 711 – 724 .e5 ( 2019 ). OpenUrl CrossRef PubMed 37. ↵ Reynaud , D. et al. SCL/TAL1 expression level regulates human hematopoietic stem cell self-renewal and engraftment . Blood 106 , 2318 – 2328 ( 2005 ). OpenUrl Abstract / FREE Full Text 38. ↵ Feng , X. et al. CTC1-STN1 terminates telomerase while STN1-TEN1 enables C-strand synthesis during telomere replication in colon cancer cells . Nat. Commun . 9 , 2827 ( 2018 ). 39. ↵ Clerici , M. , Faini , M. , Muckenfuss , L. M. , Aebersold , R. & Jinek , M . Structural basis of AAUAAA polyadenylation signal recognition by the human CPSF complex . Nat. Struct. Mol. Biol . 25 , 135 – 138 ( 2018 ). OpenUrl CrossRef PubMed 40. ↵ Mishima , Y. et al. The Hbo1-Brd1/Brpf2 complex is responsible for global acetylation of H3K14 and required for fetal liver erythropoiesis . Blood 118 , 2443 – 2453 ( 2011 ). OpenUrl Abstract / FREE Full Text 41. ↵ Le Goff , S. et al. p53 activation during ribosome biogenesis regulates normal erythroid differentiation . Blood 137 , 89 – 102 ( 2021 ). OpenUrl CrossRef PubMed 42. ↵ Lv , K. et al. CBL family E3 ubiquitin ligases control JAK2 ubiquitination and stability in hematopoietic stem cells and myeloid malignancies . Genes Dev . 31 , 1007 – 1023 ( 2017 ). OpenUrl Abstract / FREE Full Text 43. ↵ Uniacke , J. et al. An oxygen-regulated switch in the protein synthesis machinery . Nature 486 , 126 – 129 ( 2012 ). OpenUrl CrossRef PubMed Web of Science 44. ↵ Shan , J.-L. et al. Reactivating cGAS-STING Signaling by Targeting SOS1 Enhances Antitumor Immunity in NRAS-Mutant Tumors . Cancer Res . ( 2025 ). 45. ↵ An , Q. et al. Myh9 Plays an Essential Role in the Survival and Maintenance of Hematopoietic Stem/Progenitor Cells . Cells 11 , 1865 ( 2022 ). 46. ↵ Hoogeboom , R. et al. Myosin IIa Promotes Antibody Responses by Regulating B Cell Activation, Acquisition of Antigen, and Proliferation . Cell Rep . 23 , 2342 – 2353 ( 2018 ). OpenUrl PubMed 47. ↵ Li , J.-H. , Liu , S. , Zhou , H. , Qu , L.-H. & Yang , J.-H . starBase v2.0: decoding miRNA-ceRNA, miRNA-ncRNA and protein-RNA interaction networks from large-scale CLIP-Seq data . Nucleic Acids Res . 42 , D92 – 97 ( 2014 ). OpenUrl CrossRef PubMed Web of Science 48. ↵ Han , S. , Chen , X. & Huang , L . The tumor therapeutic potential of long non-coding RNA delivery and targeting . Acta Pharm. Sin. B 13 , 1371 – 1382 ( 2023 ). OpenUrl PubMed 49. ↵ Rao , T. N. et al. Attenuation of PKCδ enhances metabolic activity and promotes expansion of blood progenitors . EMBO J . 37 , e100409 ( 2018 ). OpenUrl Abstract / FREE Full Text 50. ↵ Skokowa , J. et al. LEF-1 Transcription Factor Regulates Proliferation and Differentiation of Myeloid Progenitors in Healthy Individuals and in Patients with Severe Congenital Neutropenia (CN) . Blood 106 , 390 ( 2005 ). 51. ↵ Leimbacher , P.-A. et al. MDC1 Interacts with TOPBP1 to Maintain Chromosomal Stability during Mitosis . Mol. Cell 74 , 571 – 583 .e8 ( 2019 ). OpenUrl CrossRef PubMed 52. ↵ Ma , F. et al. Systems-based identification of the Hippo pathway for promoting fibrotic mesenchymal differentiation in systemic sclerosis . Nat. Commun . 15 , 210 ( 2024 ). 53. ↵ Poulton , C. J. et al. Microcephaly with simplified gyration, epilepsy, and infantile diabetes linked to inappropriate apoptosis of neural progenitors . Am. J. Hum. Genet . 89 , 265 – 276 ( 2011 ). OpenUrl CrossRef PubMed 54. ↵ Zhang , D.-W. et al. Multiple death pathways in TNF-treated fibroblasts: RIP3-and RIP1-dependent and independent routes . Cell Res . 21 , 368 – 371 ( 2011 ). OpenUrl CrossRef PubMed Web of Science 55. ↵ Alegre , M. L. , Frauwirth , K. A. & Thompson , C. B . T-cell regulation by CD28 and CTLA-4 . Nat. Rev. Immunol . 1 , 220 – 228 ( 2001 ). OpenUrl CrossRef PubMed 56. ↵ Hedrick , S. M. , Hess Michelini , R. , Doedens , A. L. , Goldrath , A. W. & Stone , E. L . FOXO transcription factors throughout T cell biology . Nat. Rev. Immunol . 12 , 649 – 661 ( 2012 ). OpenUrl CrossRef PubMed 57. ↵ Skundric , D. S. , Cruikshank , W. W. & Drulovic , J . Role of IL-16 in CD4+ T cell-mediated regulation of relapsing multiple sclerosis . J. Neuroinflammation 12 , 78 ( 2015 ). 58. ↵ S, M., et al. Heterochronic parabiosis induces stem cell revitalization and systemic rejuvenation across aged tissues . Cell Stem Cell 29 , ( 2022 ). 59. ↵ Li , D. , Yu , W. & Lai , M . Towards understandings of serine/arginine-rich splicing factors . Acta Pharm. Sin. B 13 , 3181 – 3207 ( 2023 ). OpenUrl PubMed 60. ↵ Gerstberger , S. , Hafner , M. & Tuschl , T . A census of human RNA-binding proteins . Nat. Rev. Genet . 15 , 829 – 845 ( 2014 ). OpenUrl CrossRef PubMed 61. ↵ El , V. N. et al. A large-scale binding and functional map of human RNA-binding proteins . Nature 583 , ( 2020 ). 62. ↵ Sd , F. , L, R. & Cb , B . Quantifying negative selection in human 3’ UTRs uncovers constrained targets of RNA-binding proteins . Nat. Commun . 15 , ( 2024 ). 63. ↵ Korhonen , J. , Martinmäki , P. , Pizzi , C. , Rastas , P. & Ukkonen , E . MOODS: fast search for position weight matrix matches in DNA sequences . Bioinformatics 25 , 3181 – 3182 ( 2009 ). OpenUrl CrossRef PubMed 64. ↵ Schep , A. N. , Wu , B. , Buenrostro , J. D. & Greenleaf , W. J . chromVAR: inferring transcription-factor-associated accessibility from single-cell epigenomic data . Nat. Methods 14 , 975 – 978 ( 2017 ). OpenUrl CrossRef PubMed 65. ↵ Zhong , X. et al. RNPS1 inhibits excessive tumor necrosis factor/tumor necrosis factor receptor signaling to support hematopoiesis in mice . Proc. Natl. Acad. Sci. U. S. A . 119 , e2200128119 ( 2022 ). OpenUrl PubMed 66. ↵ Somasekharan , S. P. et al. G3BP1-linked mRNA partitioning supports selective protein synthesis in response to oxidative stress . Nucleic Acids Res . 48 , 6855 – 6873 ( 2020 ). OpenUrl CrossRef PubMed 67. ↵ Kim , S. S.-Y. , Sze , L. , Liu , C. & Lam , K.-P . The stress granule protein G3BP1 binds viral dsRNA and RIG-I to enhance interferon-β response . J. Biol. Chem . 294 , 6430 – 6438 ( 2019 ). OpenUrl Abstract / FREE Full Text 68. ↵ Clerici , M. , Faini , M. , Muckenfuss , L. M. , Aebersold , R. & Jinek , M . Structural basis of AAUAAA polyadenylation signal recognition by the human CPSF complex . Nat. Struct. Mol. Biol . 25 , 135 – 138 ( 2018 ). OpenUrl CrossRef PubMed 69. ↵ Li , L. et al. An atlas of alternative polyadenylation quantitative trait loci contributing to complex trait and disease heritability . Nat. Genet . 53 , 994 – 1005 ( 2021 ). OpenUrl CrossRef PubMed 70. ↵ Natri , H. M. et al. Cell-type-specific and disease-associated expression quantitative trait loci in the human lung . Nat. Genet . 56 , 595 – 604 ( 2024 ). OpenUrl CrossRef PubMed 71. Perez , R. K. et al. Single-cell RNA-seq reveals cell type–specific molecular and genetic associations to lupus . Science ( 2022 ) doi: 10.1126/science.abf1970 . OpenUrl CrossRef PubMed 72. ↵ Yazar , S. et al. Single-cell eQTL mapping identifies cell type–specific genetic control of autoimmune disease . Science 376 , eabf3041 ( 2022 ). 73. ↵ Korsunsky , I. et al. Fast, sensitive and accurate integration of single-cell data with Harmony . Nat. Methods 16 , 1289 – 1296 ( 2019 ). OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted August 25, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Identification and quantification of alternative polyadenylation sites in single cell RNA-seq data using scPAISO Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Identification and quantification of alternative polyadenylation sites in single cell RNA-seq data using scPAISO Yongjie Liu , Peiwen Xiong , Songyang Li , Xinjia Liu , Tao Liu , Qinglan Yang , Shuting Wu , Hongyan Peng , Yana Li , Lingling Zhang , Yafei Deng , Yong Zhu , Junping Wang , Youcai Deng bioRxiv 2025.08.20.669565; doi: https://doi.org/10.1101/2025.08.20.669565 Share This Article: Copy Citation Tools Identification and quantification of alternative polyadenylation sites in single cell RNA-seq data using scPAISO Yongjie Liu , Peiwen Xiong , Songyang Li , Xinjia Liu , Tao Liu , Qinglan Yang , Shuting Wu , Hongyan Peng , Yana Li , Lingling Zhang , Yafei Deng , Yong Zhu , Junping Wang , Youcai Deng bioRxiv 2025.08.20.669565; doi: https://doi.org/10.1101/2025.08.20.669565 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7618) Biochemistry (17635) Bioengineering (13859) Bioinformatics (41846) Biophysics (21401) Cancer Biology (18534) Cell Biology (25423) Clinical Trials (138) Developmental Biology (13352) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24286) Genetics (15582) Genomics (22463) Immunology (17700) Microbiology (40298) Molecular Biology (17141) Neuroscience (88429) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4813) Physiology (7633) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4284) Systems Biology (9808) Zoology (2267)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.