Full text
42,539 characters
· extracted from
preprint-html
· click to expand
Comparative Analysis of Pathology Foundation Models for Automated Detection of Tertiary Lymphoid Structures in H&E-Stained Digital Pathology Images | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Comparative Analysis of Pathology Foundation Models for Automated Detection of Tertiary Lymphoid Structures in H&E-Stained Digital Pathology Images Meijian Guan , Yu Sun , Merzu Belete , Anantharaman Muthuswamy , Maximilian Farma , Jenny Kaufmann , Mirna Lechpammer , Sriram Sridhar , Brandon W. Higgs , Han Si doi: https://doi.org/10.1101/2025.11.23.688074 Meijian Guan 1 Translational Data Science , Genmab, Princeton, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: megu{at}genmab.com hasi{at}genmab.com Yu Sun 2 Pathology , Genmab, Princeton, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Merzu Belete 1 Translational Data Science , Genmab, Princeton, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Anantharaman Muthuswamy 2 Pathology , Genmab, Princeton, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Maximilian Farma 3 Cornell University , Ithaca, NY, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jenny Kaufmann 4 Harvard University , Cambridge, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mirna Lechpammer 2 Pathology , Genmab, Princeton, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sriram Sridhar 1 Translational Data Science , Genmab, Princeton, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Brandon W. Higgs 1 Translational Data Science , Genmab, Princeton, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Han Si 1 Translational Data Science , Genmab, Princeton, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: megu{at}genmab.com hasi{at}genmab.com Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Tertiary lymphoid structures (TLS) have been observed in solid tumors and have been associated with better outcomes in patients treated with immunotherapy, but their dynamic nature makes identifying TLS in clinical samples challenging. Recently, pathology foundation models have emerged as powerful tools in computational pathology. In this study, we aimed to develop a computational tool capable of identifying TLS across different cancer types. To this end, we utilized multiple pathology foundation models, along with the ImageNet-pretrained ResNet50 as a baseline, to identify TLS in pancreatic ductal adenocarcinoma (PDAC) and head and neck squamous cell carcinoma (HNSCC) using hematoxylin and eosin (H&E) stained images from The Cancer Genome Atlas (TCGA) and a licensed Real-World Evidence (RWE) cohort. Both pathologist-annotated and transcriptomic signature-based TLS outcomes were employed for performance assessment. Pathologist-identified TLS-positive tumors showed higher expression of TLS signatures in both diseases. Among the models tested, PLIP and CTransPath exhibited strong performance in identifying TLS within PDAC samples (AUC = 0.94 and 0.89, respectively). However, all models struggled in the analysis of HNSCC, likely due to the increased heterogeneity of the tumor microenvironment (TME). Despite their overall utility in detecting TLS in PDAC, all foundation models demonstrated poor performance in predicting transcriptomic signature-based outcomes in both PDAC and HNSCC. This suggests that TLS signatures may reflect broader or more transient aspects of TLS biology, whereas pathology-based assessment anchors on visible morphological features, more closely aligned with the focus of foundation model training. These findings highlight the potential of advanced pathology foundation models for TLS detection and broader tumor immune profiling tasks. These models can also be utilized on routinely collected patient biosamples with nominal costs. However, further refinement is needed to enhance their utility in tumors with more complex TME. Additionally, the identification of transcriptomics-based biomarkers from H&E images remains a significant challenge, despite advancements in digital pathology. Introduction Tertiary lymphoid structures (TLS) 1 are ectopic lymphoid aggregate structures of immune cells that form within non-lymphoid tissues, including in the tumor microenvironment 2 . Found in various cancers, including head and neck squamous cell carcinoma (HNSCC) 3 and pancreatic ductal adenocarcinoma (PDAC) 4 , 5 , TLS are composed of B-cell follicles, T-cell zones and dendritic cells, and function as sites for local immune activation. TLS are crucial in cancer immunotherapy due to their role in modulating anti-tumor immune responses 6 - 8 . They act as hubs for antigen presentation, T-cell activation, and B-cell maturation, fostering a robust immune response against tumor cells. For example, TLS presence is associated with better response rates to immune checkpoint inhibitors, as they provide a localized immune environment that enhances T-cell activation and cytotoxic activity. Consequently, TLS are emerging as predictive biomarkers for immunotherapy efficacy and overall survival in cancer patients. Recent studies have highlighted the clinical significance of TLS in various cancers. A meta-analysis encompassing 17 studies with 4,291 lung cancer patients revealed that a high TLS presence correlates with improved overall survival (HR = 0.66, 95% CI: 0.50–0.88) and disease-free survival (HR = 0.46, 95% CI: 0.33–0.64) 9 . Similarly, research focusing on digestive system cancers demonstrated that the absence of TLS is associated with poorer overall survival (HR = 1.74, 95% CI: 1.50–2.03) and recurrence-free survival (HR = 1.96, 95% CI: 1.58–2.44) 10 . Furthermore, the presence of mature TLS has been linked to enhanced responses to immune checkpoint inhibitors, suggesting their potential as predictive biomarkers for immunotherapy efficacy 11 . In HNSCC, intratumoral TLS have been associated with improved patient survival and better responses to immunotherapy 12 . Similarly, research indicates that patients with PDAC who exhibit TLS tend to have a more favorable prognosis compared to those who do not 5 , 13 , 14 . Collectively, these findings highlight the prognostic and therapeutic relevance of TLS across multiple cancer types. TLS detection has traditionally relied on RNA sequencing (RNAseq) to identify specific gene signatures 3 , 15 , 16 , manual histology-based assessment to visually identify structures within tumor tissues 17 , 18 , or by detection of specific TLS-associated markers (e.g. CD3, CD20, CD23, DC-LAMP) by immunohistochemistry 6 , 18 - 20 . While informative, these methods are time-consuming, lack scalability, and often miss spatial context. Recent advancements in digital pathology and artificial intelligence (AI) have enabled the use of deep learning algorithms to detect TLS directly from hematoxylin and eosin (H&E) whole-slide images (WSIs) 21 . Published models leveraging convolutional neural networks (CNNs) have demonstrated high accuracy and scalability in automating TLS identification 22 - 24 . Despite these developments, the use of different emerging foundation models which are pre-trained architectures using large amount of histology images remains underexplored in detecting TLS. Foundation models offer potential advantages by leveraging excellent transfer learning, enabling feature extraction from diverse datasets prior to a variety of downstream tasks such as tumor detection, biomarker prediction, and survival analysis. Several pathology foundation models use self-supervised learning to enhance image analysis. CTransPath (Swin Transformer) 25 , Phikon (iBOT) 26 , Virchow (DINOv2) 27 , PLIP (Contrastive Learning) 28 , and RETCCL (Clustering-guided Contrastive Learning) 29 specialize in tasks like biomarker classification, image retrieval, demonstrating significant advancements in histopathology. In this study ( Figure 1 ), we benchmarked the performance of several commonly used pathology foundation models, including PLIP and Virchow against the pretrained model ResNet50, in identifying TLS within PDAC and HNSCC samples from The Cancer Genome Atlas (TCGA) and RWE datasets. Download figure Open in new tab Figure 1. High-level workflow for the histopathology image processing. By leveraging the strengths of these foundation models in histopathological image analysis, we aim to establish a robust methodology for objective and automated TLS identification and further assess its utility of being a prognostic biomarker associated with clinical outcomes in PDAC and HNSCC. Methods TCGA PDAC and HNSCC Cohorts TCGA provides comprehensive genomic profiles for various cancer types, including PDAC and HNSCC. We used 163 patients with high-resolution WSIs and 173 patients with RNA sequencing data, as well as their corresponding clinical information, from a diverse set of pancreatic cancer patients ( Table 1.1 ). Similarly, we selected a subset of the TCGA HNSCC cohort, consisting of detailed clinical information, histopathological images (N=100), and transcriptomic data (N=93) from head and neck cancer patients ( Table 1.2 ). In addition, this study also utilized a HNSCC cohort from a licensed RWE cohort (Tempus AI, Inc., Table 1.2 ), which comprises longitudinal data from geographically diverse oncology practices. The dataset included 188 samples profiled with histopathology images, of which 185 had matched whole-transcriptome RNA sequencing, as previously described 30 . View this table: View inline View popup Download powerpoint Table 1.1. Patient characteristics of PDAC-TCGA cohort. View this table: View inline View popup Table 1.2. Patient characteristics of HNSCC-TCGA and HNSCC-RWE cohorts. View this table: View inline View popup Download powerpoint Table 2. Summary of the evaluated pathology foundation models. TLS visual assessment on H&E whole slide images TLS visual assessment on H&E WSIs was performed by Genmab pathologists through Concentriq LS (Proscia, PA, USA). TLS was defined histologically as an organized follicular-like structure of lymphoid cells aggregation of predominant B cell lineage, presence of plasma cells and T cells. The presence of at least one dendritic cell and inclusive or adjacent high endothelial venules were also included as characteristic features of TLS on H&E images 2 , 31 . H&E Images with predominant necrosis, significant staining or tissue artefacts and from metastatic lymph node samples were not included for TLS visual assessment. TLS annotation for a subset of images was generated for visualization. TLS gene signature calculation and clustering Gene Set Variation Analysis (GSVA) 32 scores were calculated for six previously published TLS gene signatures using the normalized bulk transcriptome data (e.g., using TPM or log2-transformed CPM) from TCGA and the RWE cohort. K-means clustering was further applied to the six GSVA scores to derive TLS-high and TLS-low clusters. Uniform manifold approximation and projection (UMAP) algorithm 33 was used to visualize the clusters. Correlation between pathology annotation and TLS signature Correlation between pathologist-generated TLS label (TLS-present and TLS-absent) and TLS gene signatures-based K-means clusters (TLS-high and TLS-low) were directly evaluated to calculate the concordance score (N (TLS-present==TLS-high) + N (TLS-absent==TLS-low) )/N (Total overlapping samples) . Welch’s t-test was applied to compare TLS GSVA scores between pathologist-generated TLS-present and TLS-absent groups for both TCGA and RWE cohorts separately for both indications. Image processing Slideflow 34 package was used for image processing. Tiles were extracted from the whole slide images at 20x magnification and 299 in pixels. During the process, poor quality tiles including background tiles, blurry tiles, or tiles with high whitespace content (>60%), were discarded. Stain normalization was applied using Reinhard algorithm 35 . Feature extraction and Multiple-instance learning (ML) Five different foundation models, including RETCCL, Phikon, CTransPath, PLIP, and Virchow, as well as an ImageNet-trained ResNet50, were used to extract features from image tiles derived from each WSI ( Table 2 ). The dimensions of the image features ranged from 512 to 2560. The multi-branch clustering-constrained attention multiple instance learning (CLAM_MB) model was used to aggregate and evaluate the effectiveness of the image features. To optimize model performance, a grid search was conducted over a set of hyperparameters, including learning rate, model size, bag size, and batch size. The image set for both indications were split into 75% training and 25% testing. HNSCC TCGA and RWE images were mixed before splitting into training and testing sets. A 3-fold cross-validation method was used to on the training data to evaluate the performance of each hyperparameter combination. The best combination of hyperparameters were run three times to generate 9 data points to have a more accurate estimation. The optimized model was trained on all the training images to predict the outcome in the test data. We utilized several evaluation metrices including accuracy, specificity, recall, area under curve (AUC), and F1 score to measure the model performance. Model interpretation with attention heatmaps Attention heatmaps were generated using the best foundation model in cross-validation to highlight the most relevant regions in predicting the TLS labels. The heatmaps were then further reviewed by a pathologist to evaluate the pathological relevance of the highlighted regions in TLS identification. TLS pathology annotation of the same images was also generated before pathologists reviewed the heatmaps. Results Patient demographics A total of 173 pancreatic ductal adenocarcinoma (PDAC) patients from The Cancer Genome Atlas (TCGA) and 288 head and neck squamous cell carcinoma (HNSCC) patients from TCGA (n = 100) and an RWE cohort (n = 188) with pathology assessment were included in this study. Among these, 163 of 173 PDAC patients and 278 of 288 HNSCC patients had matched RNA sequencing (RNAseq) data. No significant differences were observed between TLS signature-based TLS-high and TLS-low groups or pathology-based TLS-present and TLS-absent groups in terms of gender, age, or pathological stage. The majority of PDAC patients were classified as stage 2 ( Table 1.1 ). Similarly, within each HNSCC cohort, no significant differences were detected in patient characteristics, including gender, age, disease stage, human papillomavirus (HPV) status, or treatment history, between TLS groups (signature-based or pathology-based). Additionally, demographic characteristics, including gender, age, and disease stage, were comparable between the HNSCC-TCGA and HNSCC-RWE cohorts ( Table 1.2 ). However, most HNSCC-TCGA patients were immunotherapy (IO) naïve, whereas the majority of HNSCC-RWE patients had received prior treatment. Furthermore, HPV positivity was lower in the HNSCC-TCGA cohort (11.24%) compared to the HNSCC-RWE cohort (60.4%). TLS label generation and evaluation We first generated GSVA scores using transcriptomic data from TCGA and the RWE cohort for the six published TLS signatures (Supplementary Table 1). A K-means clustering algorithm was applied to classify TLS-high and TLS-low groups ( Figure 2 ) for PDAC, HNSCC-TCGA, and HNSCC-RWE separately. For PDAC, we identified 84 TLS-high and 89 TLS-low slides, and for HNSCC, we identified 133 TLS-high (41 TCGA + 92 RWE) and 145 TLS-low (52 TCGA + 93 RWE) slides. Separately, pathologists performed a pathological review on H&E images to classify 91 TLS-present and 72 TLS-absent slides for PDAC, and 88 TLS-present (38 TCGA + 50 RWE) and 200 TLS-absent (62 TCGA + 138 RWE) for HNSCC. Notably, significant variations of TLS histological features between PDAC and HNSCC H&E WSIs are appreciated, such as the size, number of immune cells, cell density, maturation status, and spatial distribution in TME on H&E images. Concordance between pathology-assessed TLS levels and K-means clustering was 56.6% for PDAC and 41.0% for HNSCC (54.9% TCGA, 34.06% RWE, Supplementary Figure 1). Download figure Open in new tab Figure 2. Patient clustering with TLS signatures in PDAC-TCGA (a), HNSCC-TCGA (b), and HNSCC-RWE (c). The lower concordance rate in HNSCC was observed primarily in the RWE cohort, which exhibited greater heterogeneity in the images. To further explore the correlation between pathological review and individual TLS signatures, we conducted Welch’s t-test to compare TLS signature levels between TLS-present and TLS-absent groups in both PDAC and HNSCC (Supplementary Figure 2). In PDAC, TLS-present cases showed consistently higher TLS signature levels (p<0.05) in 5 out of 6 different TLS signatures (Supplementary Figure 2). In the HNSCC cohort, TLS-present tumors similarly displayed significantly higher TLS signature levels (p<0.05) in 5 out of 6 signatures in the HNSCC-RWE cohort. However, none of the comparisons in HNSCC-TCGA reached significance, despite showing similar trends (Supplementary Figure 2). TLS Prediction in PDAC The performance of six feature extractors—ResNet50, Phikon, RETCCL, CTransPath, PLIP, and Virchow—on predicting both RNAseq-based and pathology-derived TLS labels in the PDAC cohort is presented in Figure 3 . Overall, all models struggled to predict RNAseq-based labels, with average AUCs ranging from 0.54 to 0.65 using 3x3 cross-validation ( Figure 3.a ; Supplementary table 2). In contrast, all models performed better in predicting pathology labels with AUCs ranging from 0.71 to 0.84 with 3x3 cross-validation ( Figure 3.a ; Supplementary table 3). PLIP achieved highest AUC of 0.84 (±0.06), followed by CTransPath with an AUC of 0.78 (±0.06) on the training data ( Figure 3.a ; Supplementary table 3). This trend persisted in the testing data, where all models demonstrated improved performance on pathology labels (AUC: 0.64–0.94) compared to K-means (AUC: 0.40–0.63) ( Figure 3.b ; Table 3 ). PLIP exhibited particularly robust performance on the testing data, with an AUC of 0.94 and an F1 score of 0.95 when predicting pathology labels. CTransPath closely followed, achieving an AUC of 0.89 and an F1 score of 0.91. Surprisingly, ImageNet-trained ResNet50 also showed respectable performance, with an AUC of 0.85 and an F1 score of 0.86. The best-performing feature extractor for K-means predictions on testing data was ImageNet-pretrained ResNet50, with an AUC of 0.63 and an F1 score of 0.67 ( Figure 3.b ; Table 3 ). View this table: View inline View popup Download powerpoint Table 3. Detailed performance of foundation models in predicting TLS in PDAC test data Download figure Open in new tab Figure 3. Performance of foundation models in predicting TLS in PDAC of (a) cross validation (3x3) of training set and (b) testing set. TLS Prediction in HNSCC The same feature extractors were evaluated in the HNSCC cohort for predicting both RNAseq-based and pathology-derived TLS status ( Figure 4 ). Like PDAC, all models struggled to predict RNAseq-based labels, with average AUCs ranging from 0.63 to 0.70 in 3x3 cross-validation ( Figure 4.a ; Supplementary table 4). Most models again demonstrated better performance on pathology labels, except for ResNet50 (0.59 (±0.05)) ( Figure 4.a ; Supplementary table 5). CTransPath achieved the highest performance, with an AUC of 0.80 (±0.05), followed by Virchow and RETCCL with an AUC of 0.74 (±0.02) and 0.74 (±0.09) respectively ( Figure 4.a ; Supplementary table 5). In testing data, however, all models struggled with both pathology and K-means predictions, with Virchow achieving the highest AUC of 0.71 and an F1 score of 0.59 for pathology label predictions ( Figure 4.b ; Table 5). Notably, ResNet50 performed poorly on the HNSCC pathology label with an AUC of 0.48 and an F1 score of 0.35, a marked contrast to its robust performance in PDAC ( Figure 4 .b; Table 5). The significant drop in pathology label prediction accuracy in HNSCC compared to PDAC may be attributable to a more complex TME, greater heterogeneity in tumor location based on internal pathology assessment and previous publications 36 , 37 , and the mixed dataset from TCGA and the RWE cohort. View this table: View inline View popup Download powerpoint Table 4. Detailed performance of foundation models in predicting TLS in HNSCC test data Download figure Open in new tab Figure 4. Performance of foundation models in predicting TLS in HNSCC of (a) cross validation (3x3) of training set and (b) testing set. Model interpretation To better understand how the model detected the TLS and explore the disparities in model performance, pathology annotation and attention heatmaps were generated for both PDAC and HNSCC cohorts ( Figure 5 ). Attention heatmaps highlight regions of histopathology slides that contribute most to the model’s prediction, offering crucial interpretability in digital pathology by allowing pathologists to assess whether the AI is focusing on biologically and clinically relevant tissue features. Attention heatmaps generated by the CLAM on PLIP features for pathology-based TLS status in PDAC showed high-level of consistency with pathology annotation ( Figure 5.a ). In contrast, attention heatmaps generated with Virchow extracted features for HNSCC struggled to focus on TLS regions-only ( Figure 5.b ). Pathological evaluation indicated that the high-attention tiles identified by the model in HNSCC were influenced by pre-existing lymph nodes and non-TLS immune cell infiltration, which may exhibit visual similarities to TLS. Download figure Open in new tab Figure 5. Attention maps of PDAC (a) and HNSCC (b) slides with TLS. Discussion Self-supervised learning (SSL) has significantly advanced computational pathology by enabling the training of foundation models on large pathology image datasets. The increasing availability of publicly released models by both academic and private institutions is fostering innovation in the next generation of predictive pathology tools. Foundation models, particularly in weakly-supervised algorithms, have demonstrated superior performance and generalizability compared to traditional supervised methods. These models have been instrumental in cancer research, facilitating tumor diagnosis, biomarker prediction, and prognostic assessments 38 , 39 . Since 2022, foundation models have been integrated into weakly supervised pipelines, leading to improved diagnostic accuracy and broader applicability. As more foundation models are trained, benchmarking studies on clinically relevant tasks have also become available. Campanella et al. benchmarked eight foundation models on nine disease detection and eleven biomarker prediction tasks, finding that while newer models generally outperformed ImageNet-pretrained encoders and CTransPath, their performance varied by task and was not significantly impacted by model size 38 . Similarly, Neidlinger et al. demonstrated that model performance varied by task, with the quality and cleanliness of training data outweighing data volume or the training algorithm 39 . In this study, we evaluated the utility of the pathology foundation models for a unique task— detecting TLS presence, defined by morphological features or transcriptomic gene signatures, in two distinct indications: PDAC and HNSCC. As previously shown, detecting TLS via computational methods typically requires extensive annotation and segmentation efforts 21 - 24 . Here, we demonstrate that pathology foundation models can serve as an effective and out-of-the-box alternative for detecting morphologically defined TLS in tumor types with less complex TME. Consistent with Campanella et al. and Neidlinger et al. 38 , 39 , we also found that model performance varies by difficulty of the task while model size was not a significant factor. In general, pathology foundation models outperformed the ImageNet-pretrained ResNet50 in both the PDAC and HNSCC cohorts when predicting morphology-based TLS status. Although the foundation models consistently outperformed ResNet50 in HNSCC slide images, they all achieved lower mean AUC scores compared to their performance on PDAC tumors. This disparity is likely due to the heterogeneous tumor microenvironment of HNSCC, which may cause the foundation models to struggle in accurately identifying specific factors among multiple complex features. For example, pre-existing lymph nodes were frequently present in the HNSCC TME and may have affected the model’s performance. The largest foundation model evaluated, Virchow, did not demonstrate superior performance over smaller models, despite narrowly achieving the highest performance on the HNSCC testing data. This aligns with previous findings that model size does not necessarily correlate with performance. When predicting pathology-based TLS in PDAC, ResNet50 achieved a comparable AUC to most pathology foundation models in the training set and ranked third in the testing set. Interestingly, in the HNSCC cohort, the gap between pathology foundation models and ResNet50 widened, with foundation models consistently outperforming ResNet50 in predicting pathology-based TLS. These findings suggest that while pathology image-pretrained models still have room for improvement, they provide the most value in pathologically challenging tasks. Contrary to pathology-based label prediction, all pathology models performed poorly and did not outperform ResNet50 in identifying signature-based TLS status. This could suggest that while the foundation models excel in H&E image recognition capabilities, they struggle in RNA-based biomarker predictions, highlighting a potential limitation in their versatility across modalities. This may be partly due to the fact that TLS gene signatures are not necessarily TLS-specific and may not directly correlate with morphological features visible on H&E-stained slides. Given that current pathology foundation models largely rely on morphological features, incorporating omics data into their training could improve predictive performance, especially when target labels are defined by molecular or transcriptomic signatures 40 . Overall, model performance varied by task complexity rather than size, aligning with recent benchmarking findings. One limitation of this study is that only five foundation models were evaluated, each pretrained using different image sizes and algorithms. This selection was not intended to provide an exhaustive benchmark of all available models. Rather, the aim was to illustrate the practical utility and inherent limitations of representative models within the context of this specific application. Conclusion This study demonstrated that pathology foundation models effectively detected TLS presence in PDAC and HNSCC, outperforming ImageNet-pretrained ResNet50 in morphology-based predictions. However, foundation models struggled with transcriptomic signature-based TLS prediction, revealing cross-modal generalization challenges, and performed worse in heterogeneous HNSCC microenvironments, highlighting limitations in complex pathological contexts. Despite these challenges, pathology foundation models show strong potential as the backbone of automated digital pathology pipelines for task-specific applications. The widespread availability and standardization of H&E-stained images make them a compelling source for model deployment and scaling, particularly in biomarker discovery and patient stratification. Continued refinement will be essential to enhance their versatility and robustness across modalities and tumor heterogeneity. Data availability TCGA: https://portal.gdc.cancer.gov/ . Commercial cohort Deidentified data used in the research were collected in a real-world health care setting and are subject to controlled access for privacy and proprietary reasons. When possible, derived data supporting the findings of this study have been made available within the paper and its Supplementary Figures/Tables. Restrictions apply to the availability of additional data, which were used under license for this study. Consents and Ethics Consent This study was conducted on de-identified health information subject to an IRB exempt determination (Advarra Pro00072742) and did not involve human subject research. Author Disclosure Statement M.G., Y.S., M.B., A.M., M.L., S.S., B.H., and H.S. are current or former employees of Genmab at the time this work was completed. The authors report no additional competing interests. Funding Information This study was sponsored by Genmab. References 1. ↵ Liudahl , S.M. et al. Leukocyte Heterogeneity in Pancreatic Ductal Adenocarcinoma: Phenotypic and Spatial Features Associated with Clinical Outcome . Cancer Discov 11 , 2014 – 2031 ( 2021 ). OpenUrl Abstract / FREE Full Text 2. ↵ Schumacher , T.N. & Thommen , D.S. Tertiary lymphoid structures in cancer . Science 375 , eabf9419 ( 2022 ). OpenUrl CrossRef PubMed 3. ↵ Ruffin , A.T. et al. B cell signatures and tertiary lymphoid structures contribute to outcome in head and neck squamous cell carcinoma . Nat Commun 12 , 3349 ( 2021 ). OpenUrl CrossRef PubMed 4. ↵ Tang , R. et al. Targeting neoadjuvant chemotherapy-induced metabolic reprogramming in pancreatic cancer promotes anti-tumor immunity and chemo-response . Cell Rep Med 4 , 101234 ( 2023 ). OpenUrl PubMed 5. ↵ Zou , X. et al. Characterization of intratumoral tertiary lymphoid structures in pancreatic ductal adenocarcinoma: cellular properties and prognostic significance . J Immunother Cancer 11 ( 2023 ). 6. ↵ Sautes-Fridman , C. , Petitprez , F. , Calderaro , J. & Fridman , W.H. Tertiary lymphoid structures in the era of cancer immunotherapy . Nat Rev Cancer 19 , 307 – 325 ( 2019 ). OpenUrl CrossRef PubMed 7. Kim , H.M. & Bruno , T.C. An Introduction to Tertiary Lymphoid Structures in Cancer . Methods Mol Biol 2864 , 1 – 19 ( 2025 ). OpenUrl PubMed 8. ↵ Cabrita , R. et al. Tertiary lymphoid structures improve immunotherapy and survival in melanoma . Nature 577 , 561 – 565 ( 2020 ). OpenUrl CrossRef PubMed 9. ↵ Liu , X. , Lv , W. , Huang , D. & Cui , H. The predictive role of tertiary lymphoid structures in the prognosis and response to immunotherapy of lung cancer patients: a systematic review and meta-analysis . BMC Cancer 25 , 87 ( 2025 ). OpenUrl PubMed 10. ↵ Sun , H. et al. Prognostic value of tertiary lymphoid structures (TLS) in digestive system cancers: a systematic review and meta-analysis . BMC Cancer 23 ( 2023 ). 11. ↵ Vanhersecke , L. et al. Mature tertiary lymphoid structures predict immune checkpoint inhibitor efficacy in solid tumors independently of PD-L1 expression . Nat Cancer 2 , 794 – 802 ( 2021 ). OpenUrl PubMed 12. ↵ Liu , Z. , Meng , X. , Tang , X. , Zou , W. & He , Y. Intratumoral tertiary lymphoid structures promote patient survival and immunotherapy response in head neck squamous cell carcinoma . Cancer Immunol Immunother 72 , 1505 – 1521 ( 2023 ). OpenUrl CrossRef PubMed 13. ↵ Zhang , W.H. et al. Infiltrating pattern and prognostic value of tertiary lymphoid structures in resected non-functional pancreatic neuroendocrine tumors . J Immunother Cancer 8 ( 2020 ). 14. ↵ Hiraoka , N. et al. Intratumoral tertiary lymphoid organ is a favourable prognosticator in patients with pancreatic cancer . Br J Cancer 112 , 1782 – 1790 ( 2015 ). OpenUrl CrossRef PubMed 15. ↵ Messina , J.L. et al. 12-Chemokine gene signature identifies lymph node-like structures in melanoma: potential for patient selection for immunotherapy? Sci Rep 2 , 765 ( 2012 ). OpenUrl CrossRef PubMed 16. ↵ Petitprez , F. et al. B cells are associated with survival and immunotherapy response in sarcoma . Nature 577 , 556 – 560 ( 2020 ). OpenUrl CrossRef PubMed 17. ↵ Germain , C. et al. Presence of B cells in tertiary lymphoid structures is associated with a protective immunity in patients with lung cancer . Am J Respir Crit Care Med 189 , 832 – 844 ( 2014 ). OpenUrl CrossRef PubMed 18. ↵ Vanhersecke , L. et al. Standardized Pathology Screening of Mature Tertiary Lymphoid Structures in Cancers . Lab Invest 103 , 100063 ( 2023 ). OpenUrl CrossRef PubMed 19. Quigley , L.T. et al. Protocol for investigating tertiary lymphoid structures in human and murine fixed tissue sections using Opal-TSA multiplex immunohistochemistry . STAR Protoc 4 , 101961 ( 2023 ). OpenUrl PubMed 20. ↵ Rakaee , M. et al. Tertiary lymphoid structure score: a promising approach to refine the TNM staging in resected non-small cell lung cancer . Br J Cancer 124 , 1680 – 1689 ( 2021 ). OpenUrl CrossRef PubMed 21. ↵ Li , Z. et al. Development and Validation of a Machine Learning Model for Detection and Classification of Tertiary Lymphoid Structures in Gastrointestinal Cancers . JAMA Netw Open 6 , e2252553 ( 2023 ). OpenUrl 22. ↵ Barmpoutis , P. et al. Tertiary lymphoid structures (TLS) identification and density assessment on H&E-stained digital slides of lung cancer . PLoS One 16 , e0256907 ( 2021 ). OpenUrl CrossRef PubMed 23. van Rijthoven , M. et al. Multi-resolution deep learning characterizes tertiary lymphoid structures and their prognostic relevance in solid tumors . Commun Med (Lond) 4 , 5 ( 2024 ). OpenUrl PubMed 24. ↵ Chen , Z. et al. Deep learning on tertiary lymphoid structures in hematoxylin-eosin predicts cancer prognosis and immunotherapy response . NPJ Precis Oncol 8 , 73 ( 2024 ). OpenUrl PubMed 25. ↵ Chen , R.J. et al. Towards a general-purpose foundation model for computational pathology . Nat Med 30 , 850 – 862 ( 2024 ). OpenUrl CrossRef PubMed 26. ↵ Filiot , A. et al. Scaling Self-Supervised Learning for Histopathology with Masked Image Modeling . medRxiv ( 2023 ). 27. ↵ Vorontsov , E. et al. A foundation model for clinical-grade computational pathology and rare cancers detection . Nat Med 30 , 2924 – 2935 ( 2024 ). OpenUrl CrossRef PubMed 28. ↵ Huang , Z. , Bianchi , F. , Yuksekgonul , M. , Montine , T.J. & Zou , J. A visual-language foundation model for pathology image analysis using medical Twitter . Nat Med 29 , 2307 – 2316 ( 2023 ). OpenUrl CrossRef PubMed 29. ↵ Wang , X. et al. RetCCL: Clustering-guided contrastive learning for whole-slide image retrieval . Med Image Anal 83 , 102645 ( 2023 ). OpenUrl CrossRef PubMed 30. ↵ Kan , Z. et al. Real-world clinical multi-omics analyses reveal bifurcation of ER-independent and ER-dependent drug resistance to CDK4/6 inhibitors . Nat Commun 16 , 932 ( 2025 ). OpenUrl PubMed 31. ↵ Gago da Graca , C. , van Baarsen , L.G.M. & Mebius , R.E. Tertiary Lymphoid Structures: Diversity in Their Development, Composition, and Role . J Immunol 206 , 273 – 281 ( 2021 ). OpenUrl Abstract / FREE Full Text 32. ↵ Hanzelmann , S. , Castelo , R. & Guinney , J. GSVA: gene set variation analysis for microarray and RNA-seq data . BMC Bioinformatics 14 , 7 ( 2013 ). OpenUrl CrossRef PubMed 33. ↵ McInnes , L. , Healy , J. & Melville , J. Umap: Uniform manifold approximation and projection for dimension reduction . arXiv preprint arXiv: 1802.03426 ( 2018 ). 34. ↵ Dolezal , J.M. et al. Slideflow: deep learning for digital histopathology with real-time whole-slide visualization . BMC Bioinformatics 25 , 134 ( 2024 ). OpenUrl CrossRef PubMed 35. ↵ Reinhard , E. , Adhikhmin , M. , Gooch , B. & Shirley , P. Color transfer between images . IEEE Computer Graphics and Applications 21 , 34 – 41 ( 2001 ). OpenUrl CrossRef 36. ↵ Kane , S. et al. Pancreatic Ductal Adenocarcinoma: Characteristics of Tumor Microenvironment and Barriers to Treatment . J Adv Pract Oncol 11 , 693 – 698 ( 2020 ). OpenUrl PubMed 37. ↵ Curry , J.M. et al. Tumor microenvironment in head and neck squamous cell carcinoma . Semin Oncol 41 , 217 – 234 ( 2014 ). OpenUrl CrossRef PubMed 38. ↵ Campanella , G. et al. A clinical benchmark of public self-supervised pathology foundation models . Nat Commun 16 , 3640 ( 2025 ). OpenUrl PubMed 39. ↵ Neidlinger , P. et al. Benchmarking foundation models as feature extractors for weaklysupervised computational pathology . ArXiv abs/2408.15823 ( 2024 ). 40. ↵ Vaidya , A. et al. Molecular-driven foundation model for oncologic pathology . arXiv preprint arXiv: 2501.16652 ( 2025 ). View the discussion thread. Back to top Previous Next Posted November 26, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Comparative Analysis of Pathology Foundation Models for Automated Detection of Tertiary Lymphoid Structures in H&E-Stained Digital Pathology Images Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Comparative Analysis of Pathology Foundation Models for Automated Detection of Tertiary Lymphoid Structures in H&E-Stained Digital Pathology Images Meijian Guan , Yu Sun , Merzu Belete , Anantharaman Muthuswamy , Maximilian Farma , Jenny Kaufmann , Mirna Lechpammer , Sriram Sridhar , Brandon W. Higgs , Han Si bioRxiv 2025.11.23.688074; doi: https://doi.org/10.1101/2025.11.23.688074 Share This Article: Copy Citation Tools Comparative Analysis of Pathology Foundation Models for Automated Detection of Tertiary Lymphoid Structures in H&E-Stained Digital Pathology Images Meijian Guan , Yu Sun , Merzu Belete , Anantharaman Muthuswamy , Maximilian Farma , Jenny Kaufmann , Mirna Lechpammer , Sriram Sridhar , Brandon W. Higgs , Han Si bioRxiv 2025.11.23.688074; doi: https://doi.org/10.1101/2025.11.23.688074 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17697) Bioengineering (13894) Bioinformatics (41951) Biophysics (21455) Cancer Biology (18593) Cell Biology (25509) Clinical Trials (138) Developmental Biology (13380) Ecology (19903) Epidemiology (2067) Evolutionary Biology (24322) Genetics (15611) Genomics (22509) Immunology (17737) Microbiology (40398) Molecular Biology (17183) Neuroscience (88619) Paleontology (667) Pathology (2833) Pharmacology and Toxicology (4825) Physiology (7644) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4296) Systems Biology (9825) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.