End-to-end prediction of clinical outcomes in head and neck squamous cell carcinoma with foundation model-based multiple instance learning

preprint OA: gold CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Background Foundation models (FMs) show promise in medical AI by learning flexible features from large datasets, potentially surpassing handcrafted radiomics. Outcome prediction of head and neck squamous cell carcinomas (HNSCC) with FMs using routine imaging remains unexplored. Purpose To evaluate end-to-end FM-based multiple instance learning (MIL) for 2-year overall survival (OS), locoregional control (LRC), and freedom from distant metastasis (FFDM) prediction and risk group stratification using pretreatment CT scans in HNSCC. Materials and Methods We analyzed data of 2485 patients from three retrospective HNSCC cohorts (RADCURE, HN1, HN-PET-CT), treated between 2004 and 2017 with available pre-treatment CTs and primary gross tumor volume (GTVp) segmentations. The RADCURE cohort was split into training (n=1464) and test (N=606), with HN1 (n=131) and HN-PET-CT (n=284) as additional test cohorts. FM-based MIL models (2D, multiview and 3D) for 2-year endpoint prediction and risk stratification wre evaluated based on area under the receiver operator curve (AUROC) and Kaplan-Meier (KM) with hazard ratios (HR), compared with radiomics and assessed for multimodal enhancement with clinical baselines. Results 2D MIL models achieved 2-year test AUROCs of 0.75-0.84 (OS), 0.66-0.75 (LRC) and 0.71-0.78 (FFDM), outperforming multiview and 3D MIL (AUROCs: 0.50-0.77, p≥0.15) and comparable or superior to radiomics (AUROCs: 0.64-0.74, p≥0.012). Significant stratification was observed (HRs: 2.14-4.77, p≤0.039). Multimodal enhancement of 2-year OS/FFDM (AUROCs: 0.82-0.87, p≤0.018) was observed for patients without human papilloma virus positive (HPV+) tumors. Conclusion FM-based MIL demonstrates promise in HNSCC risk prediction, showing similar or superior performance to radiomics and enhancing clinical baselines in non-HPV+ patients. Key Results First end-to-end study using both foundation models and multiple instance learning for outcome prediction in head and neck squamous cell carcinoma. Multiple instance learning approaches predict clinically-relevant 2-year endpoints and stratify patients across external cohorts with similar or better performance than handcrafted radiomics. Multimodal inclusion of clinical and multiple instance learning information improve clinical baseline models in patients without human papillomavirus positive tumors.
Full text 58,184 characters · extracted from preprint-html · click to expand
End-to-end prediction of clinical outcomes in head and neck squamous cell carcinoma with foundation model-based multiple instance learning | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search End-to-end prediction of clinical outcomes in head and neck squamous cell carcinoma with foundation model-based multiple instance learning View ORCID Profile Asier Rabasco Meneghetti , View ORCID Profile Marta Ligero Hernández , Jens-Peter Kuehn , View ORCID Profile Steffen Löck , View ORCID Profile Zunamys Itzel Carrero , View ORCID Profile Raquel Perez-Lopez , View ORCID Profile Keno Bressem , View ORCID Profile Titus K. Brinker , View ORCID Profile Alexander T. Pearson , View ORCID Profile Daniel Truhn , View ORCID Profile Sven Nebelung , View ORCID Profile Jakob Nikolas Kather doi: https://doi.org/10.1101/2025.01.22.25320517 Asier Rabasco Meneghetti 1 Else Kroener Fresenius Center for Digital Health, Faculty of Medicine and University Hospital Carl Gustav Carus, TUD Dresden University of Technology , 01307 Dresden, Germany 2 German Cancer Consortium (DKTK), Partner site Dresden, German Cancer Research Center (DKFZ) , Heidelberg, Germany PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Asier Rabasco Meneghetti Marta Ligero Hernández 1 Else Kroener Fresenius Center for Digital Health, Faculty of Medicine and University Hospital Carl Gustav Carus, TUD Dresden University of Technology , 01307 Dresden, Germany PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Marta Ligero Hernández Jens-Peter Kuehn 3 Institute and Policlinic for Diagnostic and Interventional Radiology, Faculty of Medicine and University Hospital Carl Gustav Carus, Technische Universität Dresden , Dresden, Germany MD, MSc Find this author on Google Scholar Find this author on PubMed Search for this author on this site Steffen Löck 2 German Cancer Consortium (DKTK), Partner site Dresden, German Cancer Research Center (DKFZ) , Heidelberg, Germany 4 OncoRay – National Center for Radiation Research in Oncology, Faculty of Medicine and University Hospital Carl Gustav Carus, Technische Universität Dresden , Helmholtz-Zentrum Dresden-Rossendorf, Dresden, Germany 5 Department of Radiotherapy and Radiation Oncology, Faculty of Medicine and University Hospital Carl Gustav Carus, Technische Universität Dresden , Dresden, Germany 6 National Center for Tumor Diseases (NCT), Partner Site Dresden, Germany: German Cancer Research Center (DKFZ), Heidelberg, Germany; Faculty of Medicine and University Hospital Carl Gustav Carus, Technische Universitat Dresden ; Helmholtz-Zentrum Dresden-Rossendorf, Dresden, Germany PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Steffen Löck Zunamys Itzel Carrero 1 Else Kroener Fresenius Center for Digital Health, Faculty of Medicine and University Hospital Carl Gustav Carus, TUD Dresden University of Technology , 01307 Dresden, Germany PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Zunamys Itzel Carrero Raquel Perez-Lopez 7 Radiomics Group, Vall d’Hebron Institute of Oncology, Vall d’Hebron Barcelona Hospital Campus , Barcelona, Spain MD, MSc Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Raquel Perez-Lopez Keno Bressem 8 Department of Diagnostic and Interventional Radiology, Technical University of Munich, School of Medicine and Health, Klinikum rechts der Isar, TUM University Hospital , Ismaninger Str. 22, 81675 Munich 9 Department of Cardiovascular Radiology and Nuclear Medicine, Technical University of Munich, School of Medicine and Health, German Heart Center, TUM University Hospital , Lazarethstr. 36, 80636, Munich MD, MSc Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Keno Bressem Titus K. Brinker 10 Digital Biomarkers for Oncology Group, German Cancer Research Center (DKFZ) , INF 223, 69120 Heidelberg, Germany MD, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Titus K. Brinker Alexander T. Pearson 11 Section of Hematology/Oncology, Department of Medicine, University of Chicago , Chicago, IL, USA MD, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Alexander T. Pearson Daniel Truhn 12 Department of Diagnostic and Interventional Radiology, Medical Faculty, RWTH Aachen University , 52074 Aachen, Germany MD, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Daniel Truhn Sven Nebelung 1 Else Kroener Fresenius Center for Digital Health, Faculty of Medicine and University Hospital Carl Gustav Carus, TUD Dresden University of Technology , 01307 Dresden, Germany 12 Department of Diagnostic and Interventional Radiology, Medical Faculty, RWTH Aachen University , 52074 Aachen, Germany MD, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sven Nebelung Jakob Nikolas Kather 1 Else Kroener Fresenius Center for Digital Health, Faculty of Medicine and University Hospital Carl Gustav Carus, TUD Dresden University of Technology , 01307 Dresden, Germany 13 Department of Medicine I, University Hospital Dresden , Dresden, Germany 14 Medical Oncology, National Center for Tumor Diseases (NCT), University Hospital Heidelberg , Heidelberg, Germany MD, MSc Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jakob Nikolas Kather For correspondence: jakob.kather{at}gmail.com Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Background Foundation models (FMs) show promise in medical AI by learning flexible features from large datasets, potentially surpassing handcrafted radiomics. Outcome prediction of head and neck squamous cell carcinomas (HNSCC) with FMs using routine imaging remains unexplored. Purpose To evaluate end-to-end FM-based multiple instance learning (MIL) for 2-year overall survival (OS), locoregional control (LRC), and freedom from distant metastasis (FFDM) prediction and risk group stratification using pretreatment CT scans in HNSCC. Materials and Methods We analyzed data of 2485 patients from three retrospective HNSCC cohorts (RADCURE, HN1, HN-PET-CT), treated between 2004 and 2017 with available pre-treatment CTs and primary gross tumor volume (GTVp) segmentations. The RADCURE cohort was split into training (n=1464) and test (N=606), with HN1 (n=131) and HN-PET-CT (n=284) as additional test cohorts. FM-based MIL models (2D, multiview and 3D) for 2-year endpoint prediction and risk stratification wre evaluated based on area under the receiver operator curve (AUROC) and Kaplan-Meier (KM) with hazard ratios (HR), compared with radiomics and assessed for multimodal enhancement with clinical baselines. Results 2D MIL models achieved 2-year test AUROCs of 0.75-0.84 (OS), 0.66-0.75 (LRC) and 0.71-0.78 (FFDM), outperforming multiview and 3D MIL (AUROCs: 0.50-0.77, p≥0.15) and comparable or superior to radiomics (AUROCs: 0.64-0.74, p≥0.012). Significant stratification was observed (HRs: 2.14-4.77, p≤0.039). Multimodal enhancement of 2-year OS/FFDM (AUROCs: 0.82-0.87, p≤0.018) was observed for patients without human papilloma virus positive (HPV+) tumors. Conclusion FM-based MIL demonstrates promise in HNSCC risk prediction, showing similar or superior performance to radiomics and enhancing clinical baselines in non-HPV+ patients. Key Results First end-to-end study using both foundation models and multiple instance learning for outcome prediction in head and neck squamous cell carcinoma. Multiple instance learning approaches predict clinically-relevant 2-year endpoints and stratify patients across external cohorts with similar or better performance than handcrafted radiomics. Multimodal inclusion of clinical and multiple instance learning information improve clinical baseline models in patients without human papillomavirus positive tumors. Introduction Despite multimodal treatment, OS at 5-years remains at 30-70% for patients with HNSCC with existing prognostic methods based on the TNM staging system( 1 ). While HPV infection status is the only validated biochemical test for HNSCC, with HPV16+ oropharyngeal tumors being more radiosensitive and presenting better response to treatment( 2 ), there is no current consensus for its integration into treatment planning. Factors such as extranodal extension have been shown as having a negative prognostic effect but are not consistently quantified in a unified way( 3 ). This highlights the need for more precise prognostic biomarkers that can stratify patients into risk groups and further enable patient-tailored treatment( 4 ) such as dose de-escalation treatment for radiotherapy to reduce patient toxicity in low-risk patients( 5 ) or detection of patients with persistent tumors to consider for second-line treatment( 6 ). CT scans are routinely acquired for every patient with HNSCC and have been proposed as sources to extract quantitative, non-invasive biomarkers from. Radiomics was initially devised to extract handcrafted features from radiological images to create such image-based biomarkers( 7 ). Despite large amounts of studies and efforts for standardization( 8 ), adoption of such handcrafted radiomics biomarkers in clinical settings remains limited. One of the main drawbacks of such studies is the time-intensive requirement for manual image annotations and the variability across studies attempting to predict the same outcome( 9 ). In contrast, Deep Learning (DL) approaches offer flexibility by learning image representations without the need for prior handcrafted feature definition. DL methods have been developed for clinical applications in HNSCC to predict LRC from PET/CT images( 10 ), epithelial growth factor receptor mutation status( 11 ) or HPV infection in oropharyngeal cancer patients( 12 ) in CT. However, DL methods require larger datasets for training models than handcrafted radiomics studies, which limits the implementation of such methods in practice. An emerging area of AI research in medicine is FMs, models trained on large and unlabeled datasets to capture general patterns in the data. Then, FMs can act as backbones on downstream tasks, bypassing the need to train models from scratch. This approach allows for the extraction of general-purpose features and the development of predictive DL models from smaller datasets. Currently available FMs for extracting features that utilize radiological images for training include BioMedClip( 13 ), using 2D images of a single view or the FM by Pai et al.( 14 ) using 3D bounding boxes. Some of the current FMs require aggregation methods to analyze CT scans. Multiple instance learning (MIL) is a weakly-supervised paradigm( 15 ) that can serve as a solution by creating groups of data instances from each patient( 16 ). MIL allows for the simultaneous learning of relationships between data instances, such as different 2D slices of a CT image, and a target, such as the patient’s outcome. This is potentially advantageous as it enables handling data with ambiguous labels or identification of relevant patterns when extensive labeling of specific regions or pixels is impractical. DL-based MIL has been successfully employed within the histopathology domain for tasks such as microsatellite instability classification in colorectal cancer( 17 ) or immunotherapy response on solid tumors( 18 ). In the radiological domain, MIL has been explored( 19 , 20 ) but has not been employed before for outcome prediction using pretreatment CT images in HNSCC with FMs features. The present study aims to implement, for the first time, DL-based MIL approaches applied to FM features from pretreatment CTs of HNSCC patients for outcome prediction. As we hypothesize that MIL can aggregate FM features to learn prognostic biomarkers from the CT images, we conduct a comprehensive study including different FMs as feature extractors combined with MIL approaches to predict three different endpoints for HNSCC. The results of this study are evaluated across three different test cohorts and compared with handcrafted radiomics models and a clinical baseline. Finally, we explore correlations with clinical factors as well as the enhancement our models provide to the clinical baseline for prognosis in the overall patient population as well as non-HPV+ HNSCC patients. Materials and Methods Patient cohorts and endpoint dichotomization This retrospective study considers three cohorts with already-deidentified patients extracted from The Cancer Imaging Archive( 21 ): the RADCURE cohort( 22 ), the HN-PET-CT cohort( 23 , 24 ), and the HN1 cohort( 7 , 25 ). The RADCURE cohort consists of 3346 head and neck cancer patients treated with definitive radiotherapy or radiochemotherapy between 2005 and 2017 at the Princess Margaret Cancer Centre, Toronto, Canada. The HN-PET-CT cohort consists of 298 HNSCC patients from 4 different hospitals in Canada. These patients had received either definitive radiotherapy, radiochemotherapy or a combination of surgery and either adjuvant radiotherapy or radiochemotherapy between 2006 and 2014. The HN1 cohort consists of 137 HNSCC patients treated with radiotherapy or radiotherapy with concurrent chemotherapy at the MAASTRO clinic, Maastricht, the Netherlands starting from 2004. Patient consent was deemed unnecessary due to the public nature of the datasets. Inclusion and exclusion criteria for the study are summarized in Figure 1 . Patients who underwent surgery, had metastases upon diagnosis, or lacked proof of presenting with HNSCC through either histological proof or mention in their source publications were excluded. Inclusion criteria were defined as the availability of pretreatment CTs and GTVp segmentations. Endpoints were OS, LRC, and FFDM dichotomized at the 2-year mark similar to a prior study( 26 ) for patient retention purposes. For endpoint dichotomization, patients who were lost to follow-up before the 2-year mark or had ambiguous endpoint data, such as presenting recurrence the same day of treatment, were dropped for that endpoint. Patients with persisting disease were excluded from the 2-year LRC endpoint. OS was defined as the time from treatment start to death by any cause or loss to follow-up. LRC was defined from treatment start to local or regional tumor recurrence or loss to follow-up. FFDM was defined from treatment start until detection of distant metastasis or loss to follow-up. Download figure Open in new tab Figure 1. Flowchart showing inclusion and exclusion criteria for study cohorts as well as cohorts used for training models and for testing. Excluded patients are shown per cohort and the reason for the exclusion. Included patients were reported in previous studies ( 7 , 24 , 26 ) dealing with outcome prediction for OS ( 7 , 26 ) or LRC and FFDM ( 24 ), with either non-MIL DL or handcrafted-radiomics approaches. In contrast, the current study deals with the application of FMs as feature extractors for MIL modeling of 2-year endpoint prediction, stratification, and multimodal integration alongside clinical variables for OS, LRC and FFDM in HNSCC patients. Study design Within this retrospectively-designed study cohorts were divided into training and test cohorts to discern whether 2D, multiview or 3D FM features are best for HNSCC outcome prediction with MIL, if MIL models outperform handcrafted radiomics features and whether MIL scores add value to already-available clinical parameters ( Fig. 2A ). In order to obtain features from the CT images to be used in MIL modeling( Fig. 2B ), two approaches were considered for feature extraction: a 2D approach to extract information ( Fig. 2C ) from the three different views of the CT image (axial, coronal and sagittal) and a 3D approach based on sampling volumes within the CT ( Fig. 2D ). Download figure Open in new tab Figure 2. Workflow for overall study design, modeling and extraction. MIL models were trained to predict one of three 2-year endpoints on the training cohort and evaluated on three test cohorts (A) for each of three 2-year endpoints. For 2-year OS (B), for example, a patient would have embeddings extracted from the CT which would be used as input into a MIL transformer, learning simultaneously on embeddings from all CTs slices/subvolumes and pooling them to return patient-level scores for 2-year OS. Embeddings are extracted from FMs using either (C) axial or multiple views of CT slices containing the GTVp or from (D) subvolumes sampled around the GTVp center for a 3D approach. Preprocessing and feature extraction For all CT images, the patient body was segmented through TotalSegmentator( 27 ) and used to mask out the patient bed. For 2D MIL, for each CT view all 2D CT slices containing the GTVp mask were selected. CT images and GTVp segmentations were extracted and resampled to voxel spacing of 1×1×1 mm³ through the SimpleITK Python library. Hounsfield Units (HU) were clipped with a soft tissue windowing approach with a level of 50 HU and a window width of 120 HU. Images were converted into grayscale, cropped on the GTVp center of mass to size 224×224 pixels and normalized to FM preprocessing parameters (Supplementary Table 1). Different pretrained feature extractors were considered to obtain slice-level representations of the patient: i) ImageNet-pretrained ResNet50, ii) ImageNet-pretrained SwinViT( 28 ), and iii) the PMC-15 pretrained vision encoder extracted from the BioMedClip FM( 13 ). 2D models were constructed using either axial view features or features from all three CT views (multiview models). In 3D MIL, for the FM by Pai et al.( 14 ) images were preprocessed according to FM specifications of the original publication: values normalized to a range of −1024 and 2048 HU and resampled to 1×1×1 mm³. To capture different contexts of the tumor and its surrounding tissue, 50×50×50 voxel volumes were created by randomly sampling 100 times from a multivariate normal distribution with variance of 16 and with the coordinates of the center of mass of the GTVp as mean. An ImageNet-pretrained inflated 3D ResNet50( 29 ) was also considered, with features extracted similarly to the previously mentioned FM and using the same HU windowing. MIL model training Extracted features were input into a MIL Transformer ( 30 , 31 ) using the STAMP package version 1.0.1. A MIL model was trained for each endpoint through class-stratified 5-fold cross-validation. For each MIL model and endpoint, the model with the area under the receiver operator curve (AUROC) closest to the median AUROC value of the cross-validation holdout set was chosen for deployment in the test cohorts for classification and stratification purposes. Subgroup analysis to assess performance changes based on contrast enhancement status were also conducted. Training specifications are shown in the Supplementary Materials. Handcrafted radiomics extraction and modeling Morphological, statistical, and texture features were extracted alongside from the base image using PyRadiomics with only IBSI-compliant features. Feature selection consisted of variance filtering, z-score standardization, hierarchical clustering with correlation distance cutoff of 0.20 for cluster membership with average Spearman correlation for cluster representative selection followed by minimum redundancy maximum relevance selection similar to another study for HNSCC handcrafted radiomics modelling( 32 ). The number of features selected was the number of samples in the minority class divided by 10. Selected features were then input into a logistic regression model for training and tested on the test cohorts. Subgroup analysis to assess performance changes based on contrast enhancement status were also conducted. Extraction parameters are shown in Supplementary Table 2. Clinical baseline and multimodal enhancement Baseline clinical models were developed using the following factors: T stage, N stage, HPV status, age, sex and treatment using a logistic regression. Relationships between MIL scores and clinical baseline features were assessed by the coefficients of a multivariate logistic regression model containing the MIL scores and the clinical baseline features on the training cohort as well as with a correlation analysis between MIL scores and clinical baseline factors in the external test cohorts. Enhancement of model performance by inclusion of MIL scores was assessed in all patients of the test cohorts as well as in the subgroup of patients without HPV+ tumors. Statistical analysis Statistical comparisons between categorical variables in the training and test cohorts were conducted through the chi-squared test or the exact Fisher test. Continuous variables were compared through the Mann-Whitney U-test. The AUROC was chosen as a summary metric of classification performance for 2-year endpoints. AUROC 95% confidence intervals (95% CI) were obtained from the empirical distribution through 1000 class-stratified bootstraps. Risk group stratification was conducted through the KM estimator using the 2-year MIL scores as risk scores. The cutoff was the median score on the training cohort, and risk group separation was assessed through the Cox HR and the logrank test p-value. For explainability, attention scores were generated and the top-3 CT slices from selected patients were displayed with element-wise gradcam used to visualize the regions important for model-prediction within the slice through the gradcam Python library. AUROC differences were assessed through score permutation tests: for 1000 bootstraps, predictions from randomly selected patients were permuted between models and the resulting AUROCs were used to define a null distribution and heuristic p-value. All statistical tests were conducted in a two-sided manner unless specified. Significance of MIL scores in multivariate logistic regression was assessed through significance of log-odds ratios (LOR). Correlations of clinical baseline factors with MIL scores were assessed through Spearman rho or through point-biserial correlation. A p-value threshold of 0.05 was chosen for significance. Multiple comparison adjustment was performed through the Benjamini-Hochberg method. All analyses were conducted in Python 3.11 or R version 4.4.0. Code available at: https://github.com/KatherLab/HNSCC_FM_MIL_study . Results Patient cohort clinical characteristics The characteristics of the patients in the RADCURE train (N=1404), RADCURE test (N=606), HN1 (N=131), and HN-PET-CT (N=284) cohorts are shown in Table 1 . The RADCURE test cohort patients had significantly lower ECOG stages, higher N-stages, and a higher proportion of oropharyngeal tumors (all p<0.001). The HN1 cohort had a significantly higher proportion of patients with recurrence before 2 years, higher ECOG status, and significantly fewer HPV+ patients (all p<0.001). Tumors in the HN-PET-CT cohort had significantly larger volume, a higher proportion of patients treated with chemotherapy and significantly higher 2-year OS (all p<0.001). CT scanner parameters across cohorts are shown in Supplementary Table 3. No statistically significant differences in age, sex, radiation dose, number of fractions, and 2-year FFDM were observed across cohorts. View this table: View inline View popup Table 1. Demographic, clinical and endpoint data for included patients of the training and test cohorts. Continuous variables are indicated by median [interquartile range]. Categorical variables are indicated with their number and percentage per category. Statical comparisons between training and each test cohort are also shown. Missing categories/variables within one of the cohorts are indicated by NA. T stage, N stage and Staging were acquired following the American Joint Committee on Cancer 7th Edition criteria. Number of patients dropped for each 2-year dichotomised prognostic endpoint are labelled as Dropout. Performance of the clinical baseline model First, we evaluated the predictive performance of the clinical baseline models (Suppl. Tables 4-6). For 2-year OS the model showed AUROCs of 0.83 [0.78-0.88], 0.82 [0.73-0.89] and 0.78 [0.70-0.86] for the RADCURE test, HN1 and HN-PET-CT cohorts. For 2-year LRC, the model showed AUROCs of 0.81 [0.76-0.86], 0.79 [0.69-0.86] and 0.74 [0.66-0.82] for RADCURE test, HN1 and HN-PET-CT respectively. Finally, the model showed AUROCs of 0.76 [0.70.0.82], 0.86 [0.78-0.93] and 0.74 [0.66-0.81] for the RADCURE test, HN1 and HN-PET-CT cohorts for 2-year FFDM. Performance of the handcrafted radiomics model Secondly, we assessed the performance of the handcrafted radiomics models (Suppl. Tables 7-9). For 2-year OS the AUROCs were of 0.73 [0.68-0.80], 0.73 [0.64-0.85] and 0.67 [0.56-0.79] for the RADCURE-test, HN1 and HN-PET-CT cohorts. For 2-year LRC, AUROCs were 0.68 [0.62-0.76] for RADCURE-test, 0.73 [0.60-0.84] for HN1 and 0.63 [0.52-0.74] for HN-PET-CT. For 2-year FFDM, AUROCs were 0.71 [0.64-0.77], 0.68 [0.52-0.83] and 0.67 [0.56-0.79] for RADCURE test, HN1 and HN-PET-CT respectively. Models trained using the subgroups of non-CE (Suppl. Tables 10-12) and CE patients (Suppl. Tables 13-15) did not show significantly higher performance (p≥0.20) when compared to the models trained on the entire training cohort (Suppl. Table 16). Deep learning models classify 2-year events for OS, LRC and FFDM For MIL models, performances for the best models per model type and endpoint are shown in Figure 3 alongside handcrafted radiomics and clinical models. Multiview and 3D models’ performances were not significantly higher than 2D MIL models with axial features (p≥0.15, Suppl. Table 17). For MIL models with 2D axial features, the best model for 2-year OS showed AUROCs of 0.75 [0.69-0.80], 0.77 [0.68-0.85] and 0.84 [0.77-0.90] for the RADCURE test, HN1 and HN-PET-CT cohorts respectively. For 2-year LRC, AUROCs of 0.75 [0.68-0.81], 0.66 [0.52-0.78] and 0.72 [0.61-0.80] were obtained for the RADCURE test, HN1 and HN-PET-CT cohorts respectively. Finally, for 2-year FFDM, the best-performing model achieved AUROCs of 0.78 [0.71-0.84], 0.77 [0.58-0.93] and 0.73 [0.61-0.83] for each test cohort. ROC curves across all extractors and endpoints are shown in Suppl. Fig. 1-3. Contrast subgroup-specific models did not significantly outperform models trained on the entire training cohort (p≥0.20, Suppl.Table 18). When compared against handcrafted radiomics models as shown in Table 2 , 2D MIL models showed similar (p≥0.15) performances except for 2-year OS in the HN-PET-CT cohort where 2D MIL significantly outperformed handcrafted radiomics (p=0.012). Download figure Open in new tab Figure 3. Performance summary for 2-year endpoint classification for the clinical baseline, handcrafted radiomics and the best 2D axial, multiview and 3D MIL models shown across the three test cohorts. AUROC distributions from 1000 class-stratified bootstraps are shown for (A) 2-year OS, (B) 2-year LRC and (C) 2-year FFDM. View this table: View inline View popup Download powerpoint Table 2. AUROC across 2-year endpoints and cohort for the obtained radiomics models alongside p-value of one-sided permutation tests with alternative hypothesis of superiority of MIL models over radiomics models. Deep learning MIL scores stratify patients for time-to-event analysis Next, we investigated whether MIL models could stratify the full time-to-event endpoints. For the best 2D axial MIL models, OS showed significant stratification (p<0.001) across all three test cohorts as shown in Figure 4 . For LRC and FFDM (Suppl. Fig. 4), the 2D axial MIL models were able to significantly stratify all (p≤0.039) and two out of three cohorts (p≤0.032), respectively, tending to show higher HRs and more significant stratification when compared to handcrafted-radiomics (p≤0.20) for OS but not higher than clinical baselines as seen in Table 3 . Multiview and 3D models did not show better stratification than 2D MIL models (Suppl. Fig. 5-6). KM curves for clinical baseline and handcrafted radiomics models are shown in Suppl. Fig. 7-8. Download figure Open in new tab Figure 4. KM plots for OS stratification with the best 2D MIL model for the (A) RADCURE test, (B) HN1 and (C) HN-PET-CT cohorts alongside exemplary axial CT slices with heatmaps for the three CT slices with highest attention belonging to a high-risk patient for RADCURE test (D), HN1 (E) and HN-PET-CT (F) are shown alongside patient characteristics. CT scans are shown after image preprocessing and with a green contour indicating the GTVp segmentation. View this table: View inline View popup Download powerpoint Table 3. KM stratification summary with HR [95% CI] between high and low-risk groups across all test cohorts for 2D axial MIL, radiomics and clinical models alongside logrank p-value and cutoff extracted from the training cohort. NAs indicate that stratification into two groups produced only one group with events and was thus not possible to assess separation. HRs smaller than 1 indicate that patients with higher risk scores have a smaller probability of event. Deep learning MIL scores are significant predictors in multivariate analysis When adjusted for clinical baseline factors in a multivariate setting ( Fig. 5A-C ) 2D MIL scores remained significant predictors (p≤0.001), with increases in MIL scores resulting in higher probability of presenting 2-year events for all endpoints (Suppl. Tables 19-21) For 2-year OS, the 2D MIL clinically-adjusted LOR was 0.98 [0.82-1.05], for 2-year LRC LOR was 0.89 [0.77-1.01] and for 2-year FFDM LOR was 0.72 [0.56-0.87]. When examining correlations with clinical factors in the external test cohorts ( Fig. 5D-F ), MIL scores consistently showed mild to strong correlation with higher T stages for 2-year OS (rho:0.26-0.55), 2-year LRC (rho:0.22-0.51) and 2-year FFDM (rho:0.22-0.65) and consistently showed mild to moderate correlation with N stage for 2-year FFDM (rho:0.32-0.56). Download figure Open in new tab Figure 5. Forest plots for multivariate logistic regression with 2D MIL scores in training (A, B,C) and correlation analysis for test cohorts (D,E,F) per endpoint. Coefficients of multivariate logistic regression (log-odds ratio) are shown for 2-year OS (A), 2-year LRC (B) and 2-year FFDM (C) alongside significance level per coefficient and factoring of variables for analysis. Correlation coefficients of clinical factors through Spearman rho (Age, T stage, N stage) and point-biserial correlation (Treatment, HPV status and gender) are shown for all three test cohorts for 2-year OS (D), 2-year LRC (E) and 2-year FFDM (F) with positive coefficients indicating a positive association between ordering of the clinical factors and increasing 2D MIL scores. Significance levels:.: 0.05<p<0.10, *: 0.01<p<0.05, **:0.001<p<0.01, ***: p<0.001. Multimodal input improves model performance Finally, we investigated multimodal performance enhancement through combining the clinical baseline and 2D MIL scores. When tested on the external test sets, as shown in Figure 6 , while models combining clinical baseline features and MIL scores tended to display similar performances compared to the clinical baseline model across endpoints, significant performance improvements were observed ( Fig. 6A-C ) for 2-year OS in the HN-PET-CT cohort (0.88 [0.81-0.93] vs 0.78 [0.70-0.86], p=0.018). For the subgroup analysis in non-HPV+ patients ( Fig. 6D-F ), significantly higher performances were observed in the HN-PET-CT cohort for 2-year OS (0.87 [0.79-0.92] vs 0.75 [0.63-0.83], p=0.018) and in the RADCURE test cohort for 2-year FFDM (0.82 [0.74-0.87] vs 0.77 [0.69-0.85], p=0.006). Download figure Open in new tab Figure 6. Performance and statistical comparison summary for 2-year endpoint classification for all patients ( A,B,C) and patients without HPV-positive HNSCC (D,E,F) in the test cohorts for the clinical baseline, 2D MIL and combined multivariate model with clinical baseline and 2D MIL scores. AUROC distributions from 1000 class-stratified bootstraps are shown across all three test cohorts for (A,D) 2-year OS, (B,E) 2-year LRC and (C,F) 2-year FFDM. Significance levels: ns: non-significant,.:0.05<p<0.10, *: 0.01<p<0.05, **:0.001<p<0.01, ***: p<0.001. Discussion FMs are an emerging area of research for medical AI and have shown great potential as very flexible feature extractors for a variety of downstream tasks in domains such as histopathology( 33 ). Our study assesses, for the first time, FMs as feature extractors with MIL for endpoint prediction in HNSCC using pretreatment CT images. While approaches similar to MIL with DL are not unknown in the radiological domain, such as with brain MRI for Alzheimer’s disease prediction with multiple Transformers( 34 ), such studies do not deal with cancer prognosis and such model training approaches need to train to extract features from the raw images, whereas our study leverages FMs to sidestep that part of training, focusing only on downstream tasks alongside comparing such extractors to both handcrafted-radiomics and clinical baselines within a multicentric external validation setting. We have shown that 2D MIL models with FM feature extraction achieve good classification performance across 2-year endpoints for HNSCC patients with performances similar to the best DL image models in previous RADCURE studies( 26 ) for 2-year OS as well as significant stratification of patients into risk groups, indicating 2D MIL approaches with FMs as competitive for 2-year endpoint prediction as well as stratification of patients into risk groups for OS and LRC, with broader stratification for LRC than other studies employing 2D approaches for HNSCC( 35 ). Information from multiple views of the CT did not result in consistent increases in performance or stratification, suggesting that information from coronal and sagittal views is of limited relevance when already considering the axial slices for outcome prediction. When considering the 3D FM by Pai et al.( 14 ), performances across all 2-year endpoints were lower than in 2D models. The cause of this might be both a lower quantity of data used( 36 ) for originally training the 3D FM when compared to other FMs like BioMedClip as well as training mostly on general abdominal and chest lesions, suggesting that FMs trained on specific sections of the body may not be optimal for prediction within other body regions. Thus, there is a need for larger CT datasets to create better-performing FMs as well as for assessment of FMs in clinically-relevant prediction tasks inside and outside of their original anatomical region to better define their applicability scope. Handcrafted radiomics achieved generally similar or slightly lower performance for 2-year endpoint prediction and stratification compared to 2D MIL models. This performance disparity is in accordance with Gouthamchand et al.( 37 ) where handcrafted radiomics models were shown to systematically offer slightly lower performances over DL models in HNSCC studies. The inclusion of information of not just the GTVp but also of surrounding context from the CT slice through the DL feature representations might provide information related to aspects such as tumor spread across the slice and explain the observed gains in both classification performance and stratification. While previous studies have included handcrafted radiomics information outside the GTVp( 38 ), showcasing increased performance compared to GTVp-only handcrafted radiomics features, these studies require tuning the size of peritumoral volume to be included in contrast to DL approaches. While statistically significant predictors in a multivariate setting, MIL scores showed positive correlation across endpoints with tumor T stage for all external cohorts; indicating that models learned patterns associated with known clinical factors. Correlations of scores to clinical parameters were shown to be endpoint-dependent as positive correlations with N-stage for 2-year FFDM were observed across all three test cohorts, which is a known risk factor for distant metastasis in HNSCC( 39 ) while showing little to no correlations to N-stage for 2-year OS MIL scores. Increases in performance for the multimodal models when compared to the clinical baseline were observed in the subset of patients of non-HPV+ tumors, indicating potential for MIL approaches for multimodal modeling with clinical baselines. Such increases were, however, not consistent across all endpoints and test cohorts. Bias mitigation techniques for learning scores that are less redundant with known clinical factors while training for endpoint-prediction tasks( 40 ) could be considered for future work in order to better synergize AI predictions with known clinical baselines. The present study is not without its limitations; considered data is from publicly.available retrospective cohorts and further validation is required to extend the modeling scope beyond a proof of concept for the applicability of MIL and FM approaches in HNSCC with pretreatment CTs. Secondly, the paper does not consider handcrafted radiomics features within the peritumoral area. Thirdly, while other 3D FMs based on CT exist such as MERLIN( 41 ) or CT-CLIP( 42 ), they are not evaluated due to non-suitability for MIL modelling. And lastly, the study does not explore applicability of FMs as extractors in settings with low training data for endpoint prediction models. In conclusion, the study has shown the potential of 2D MIL models with FM backbones for competitive endpoint prediction and stratification in HNSCC patients with similar or higher performance than handcrafted radiomics models as well as added performance for non-HPV+ HNSCC tumors. Data Availability Code to reproduce the results is available at https://github.com/KatherLab/HNSCC_FM_MIL_study , data used for development and validation of the paper can be accessed through request to The Cancer Imaging Archive (TCIA) Funding Disclosure and COI JNK is supported by the German Cancer Aid (DECADE, 70115166), the German Federal Ministry of Education and Research (PEARL, 01KD2104C; CAMINO, 01EO2101; SWAG, 01KD2215A; TRANSFORM LIVER, 031L0312A; TANGERINE, 01KT2302 through ERA-NET Transcan; Come2Data, 16DKZ2044A; DEEP-HCC, 031L0315A), the German Academic Exchange Service (SECAI, 57616814), the German Federal Joint Committee (TransplantKI, 01VSF21048) the European Union’s Horizon Europe and innovation programme (ODELIA, 101057091; GENIAL, 101096312), the European Research Council (ERC; NADIR, 101114631), the National Institutes of Health (EPICO, R01 CA263318) and the National Institute for Health and Care Research (NIHR, NIHR203331) Leeds Biomedical Research Centre. KKB is supported by the European Union’s Horizon Europe and innovation programme (COMFORT, 101079894), Bayern Innovativ and Wilhelm Sander Foundation. DT is supported by the German Federal Ministry of Education (TRANSFORM LIVER, 031L0312A; SWAG, 01KD2215B), Deutsche Forschungsgemeinschaft (DFG) (TR 1700/7-1), and the European Union (Horizon Europe, ODELIA, GA 101057091). DT received honoraria for lectures by Bayer, GE, and Philips and holds shares in StratifAI GmbH, Germany and in Synagen GmbH, Germany. The views expressed are those of the author(s) and not necessarily those of the NHS, the NIHR or the Department of Health and Social Care. This work was funded by the European Union. List of Abbreviations FM foundation model MIL multiple instance learning HNSCC head and neck squamous cell carcinoma GTVp primary gross tumor volume AUROC area under the receiver operator curve LOR log-odds ratio HR hazard ratio OS overall survival LRC locoregional control FFDM freedom from distant metastasis HPV+ human papillomavirus positive References 1. ↵ Johnson DE , Burtness B , Leemans CR , Lui VWY , Bauman JE , Grandis JR . Head and neck squamous cell carcinoma . Nat Rev Dis Primer . 2020 ; 6 ( 1 ): 92 . doi: 10.1038/s41572-020-00224-3 . OpenUrl CrossRef PubMed 2. ↵ Ang KK , Harris J , Wheeler R , et al. Human Papillomavirus and Survival of Patients with Oropharyngeal Cancer . N Engl J Med. Massachusetts Medical Society ; 2010 ; 363 ( 1 ): 24 – 35 . doi: 10.1056/NEJMoa0912217 . OpenUrl CrossRef PubMed Web of Science 3. ↵ Huang SH , Chernock R , O’Sullivan B , Fakhry C . Assessment Criteria and Clinical Implications of Extranodal Extension in Head and Neck Cancer . Am Soc Clin Oncol Educ Book. Wolters Kluwer ; 2021 ;( 41 ): 265 – 278 . doi: 10.1200/EDBK_320939 . OpenUrl CrossRef 4. ↵ Baumann M , Krause M , Overgaard J , et al. Radiation oncology in the era of precision medicine . Nat Rev Cancer . 2016 ; 16 ( 4 ): 234 – 249 . doi: 10.1038/nrc.2016.18 . OpenUrl CrossRef PubMed 5. ↵ Andreassen CN , Eriksen JG , Jensen K , et al. IMRT – Biomarkers for dose escalation, dose de-escalation and personalized medicine in radiotherapy for head and neck cancer . Oral Oncol . 2018 ; 86 : 91 – 99 . doi: 10.1016/j.oraloncology.2018.09.001 . OpenUrl CrossRef PubMed 6. ↵ Steinbichler TB , Lichtenecker M , Anegg M , et al. Persistent Head and Neck Cancer Following First-Line Treatment . Cancers . 2018 ; 10 ( 11 ): 421 . doi: 10.3390/cancers10110421 . OpenUrl CrossRef PubMed 7. ↵ Aerts HJWL , Velazquez ER , Leijenaar RTH , et al. Decoding tumour phenotype by noninvasive imaging using a quantitative radiomics approach . Nat Commun. Nature Publishing Group ; 2014 ; 5 ( 1 ): 4006 . doi: 10.1038/ncomms5006 . OpenUrl CrossRef PubMed 8. ↵ Zwanenburg A , Vallières M , Abdalah MA , et al. The Image Biomarker Standardization Initiative: Standardized Quantitative Radiomics for High-Throughput Image-based Phenotyping . Radiology. Radiological Society of North America ; 2020 ; 295 ( 2 ): 328 – 338 . doi: 10.1148/radiol.2020191145 . OpenUrl CrossRef PubMed 9. ↵ Ligero M , Gielen B , Navarro V , et al. A whirl of radiomics-based biomarkers in cancer immunotherapy, why is large scale validation still lacking? Npj Precis Oncol. Nature Publishing Group ; 2024 ; 8 ( 1 ): 1 – 9 . doi: 10.1038/s41698-024-00534-9 . OpenUrl CrossRef 10. ↵ Starke S , Zwanenburg A , Leger K , et al. Multitask Learning with Convolutional Neural Networks and Vision Transformers Can Improve Outcome Prediction for Head and Neck Cancer Patients . Cancers . 2023 ; 15 ( 19 ): 4897 . doi: 10.3390/cancers15194897 . OpenUrl CrossRef PubMed 11. ↵ Zheng Y-M , Pang J , Liu Z-J , et al. A CT-based Deep Learning Radiomics Nomogram for the Prediction of EGFR Mutation Status in Head and Neck Squamous Cell Carcinoma . Acad Radiol . 2024 ; 31 ( 2 ): 628 – 638 . doi: 10.1016/j.acra.2023.06.026 . OpenUrl CrossRef PubMed 12. ↵ Lang DM , Peeken JC , Combs SE , Wilkens JJ , Bartzsch S . Deep Learning Based HPV Status Prediction for Oropharyngeal Cancer Patients . Cancers . 2021 ; 13 ( 4 ): 786 . doi: 10.3390/cancers13040786 . OpenUrl CrossRef PubMed 13. ↵ Zhang S , Xu Y , Usuyama N , et al. BiomedCLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs . arXiv; 2024 . doi: 10.48550/arXiv.2303.00915 . OpenUrl CrossRef 14. ↵ Pai S , Bontempi D , Hadzic I , et al. Foundation model for cancer imaging biomarkers . Nat Mach Intell. Nature Publishing Group ; 2024 ; 6 ( 3 ): 354 – 367 . doi: 10.1038/s42256-024-00807-9 . OpenUrl CrossRef PubMed 15. ↵ Misera L , Müller-Franzes G , Truhn D , Kather JN . Weakly Supervised Deep Learning in Radiology . Radiology. Radiological Society of North America ; 2024 ; doi: 10.1148/radiol.232085 . OpenUrl CrossRef 16. ↵ Carbonneau M-A , Cheplygina V , Granger E , Gagnon G . Multiple instance learning: A survey of problem characteristics and applications . Pattern Recognit . 2018 ; 77 : 329 – 353 . doi: 10.1016/j.patcog.2017.10.009 . OpenUrl CrossRef 17. ↵ Kiehl L , Kuntz S , Höhn J , et al. Deep learning can predict lymph node status directly from histology in colorectal cancer . Eur J Cancer. Elsevier ; 2021 ; 157 : 464 – 473 . doi: 10.1016/j.ejca.2021.08.039 . OpenUrl CrossRef 18. ↵ Ligero M , Serna G , El Nahhas OSM , et al. Weakly Supervised Deep Learning Predicts Immunotherapy Response in Solid Tumors Based on PD-L1 Expression . Cancer Res Commun . 2024 ; 4 ( 1 ): 92 – 102 . doi: 10.1158/2767-9764.CRC-23-0287 . OpenUrl CrossRef PubMed 19. ↵ Chang R , Qi S , Wu Y , et al. Deep multiple instance learning for predicting chemotherapy response in non-small cell lung cancer using pretreatment CT images . Sci Rep. Nature Publishing Group ; 2022 ; 12 ( 1 ): 19829 . doi: 10.1038/s41598-022-24278-3 . OpenUrl CrossRef 20. ↵ López-Pérez M , Schmidt A , Wu Y , Molina R , Katsaggelos AK . Deep Gaussian processes for multiple instance learning: Application to CT intracranial hemorrhage detection . Comput Methods Programs Biomed . 2022 ; 219 : 106783 . doi: 10.1016/j.cmpb.2022.106783 . OpenUrl CrossRef 21. ↵ Clark K , Vendt B , Smith K , et al. The Cancer Imaging Archive (TCIA): Maintaining and Operating a Public Information Repository . J Digit Imaging . 2013 ; 26 ( 6 ): 1045 – 1057 . doi: 10.1007/s10278-013-9622-7 . OpenUrl CrossRef PubMed Web of Science 22. ↵ Welch ML , Kim S , Hope A , et al. Computed Tomography Images from Large Head and Neck Cohort (RADCURE) . The Cancer Imaging Archive (TCIA ); 2023 . doi: 10.7937/J47W-NM11 . OpenUrl CrossRef 23. ↵ Vallières M , Kay-Rivest E , Jean Perrin L , et al. Data from Head-Neck-PET-CT . The Cancer Imaging Archive (TCIA ); doi: 10.7937/K9/TCIA.2017.8oje5q00 . OpenUrl CrossRef 24. ↵ Vallières M , Kay-Rivest E , Perrin LJ , et al. Radiomics strategies for risk assessment of tumour failure in head-and-neck cancer . Sci Rep. Nature Publishing Group ; 2017 ; 7 ( 1 ): 10117 . doi: 10.1038/s41598-017-10371-5 . OpenUrl CrossRef PubMed 25. ↵ Wee L , Dekker A . Data from HEAD-NECK-RADIOMICS-HN1 . The Cancer Imaging Archive (TCIA ); doi: 10.7937/tcia.2019.8kap372n . OpenUrl CrossRef 26. ↵ Kazmierski M , Welch M , Kim S , et al. Multi-institutional Prognostic Modeling in Head and Neck Cancer: Evaluating Impact and Generalizability of Deep Learning and Radiomics . Cancer Res Commun . 2023 ; 3 ( 6 ): 1140 – 1151 . doi: 10.1158/2767-9764.CRC-22-0152 . OpenUrl CrossRef 27. ↵ Wasserthal J , Breit H-C , Meyer MT , et al. TotalSegmentator: Robust Segmentation of 104 Anatomic Structures in CT Images . Radiol Artif Intell. Radiological Society of North America ; 2023 ; 5 ( 5 ): e230024 . doi: 10.1148/ryai.230024 . OpenUrl CrossRef 28. ↵ Liu Z , Lin Y , Cao Y , et al. Swin Transformer: Hierarchical Vision Transformer using Shifted Windows . IEEE Computer Society ; 2021 . p. 9992 – 10002 . doi: 10.1109/ICCV48922.2021.00986 . OpenUrl CrossRef 29. ↵ Carreira J , Zisserman A. Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset. arXiv ; 2018 . doi: 10.48550/arXiv.1705.07750 . OpenUrl CrossRef 30. ↵ Wagner SJ , Reisenbüchler D , West NP , et al. Transformer-based biomarker prediction from colorectal cancer histology: A large-scale multicentric study . Cancer Cell . 2023 ; 41 ( 9 ): 1650 – 1661 .e4. doi: 10.1016/j.ccell.2023.08.002 . OpenUrl CrossRef PubMed 31. ↵ Nahhas OSME , van Treeck M , Wölflein G , et al. From Whole-slide Image to Biomarker Prediction: A Protocol for End-to-End Deep Learning in Computational Pathology . arXiv; 2023 . doi: 10.48550/arXiv.2312.10944 . OpenUrl CrossRef 32. ↵ Rabasco Meneghetti A , Zwanenburg A , Leger S , et al. Definition and validation of a radiomics signature for loco-regional tumour control in patients with locally advanced head and neck squamous cell carcinoma . Clin Transl Radiat Oncol . 2021 ; 26 : 62 – 70 . doi: 10.1016/j.ctro.2020.11.011 . OpenUrl CrossRef 33. ↵ Chen RJ , Ding T , Lu MY , et al. Towards a general-purpose foundation model for computational pathology . Nat Med. Nature Publishing Group ; 2024 ; 30 ( 3 ): 850 – 862 . doi: 10.1038/s41591-024-02857-3 . OpenUrl CrossRef 34. ↵ Jang J , Hwang D. M3T: three-dimensional Medical image classifier using Multi-plane and Multi-slice Transformer . 2022 IEEECVF Conf Comput Vis Pattern Recognit CVPR . 2022 . p. 20686 – 20697 . doi: 10.1109/CVPR52688.2022.02006 . OpenUrl CrossRef 35. ↵ Starke S , Leger S , Zwanenburg A , et al. 2D and 3D convolutional neural networks for outcome modelling of locally advanced head and neck squamous cell carcinoma . Sci Rep. Nature Publishing Group ; 2020 ; 10 ( 1 ): 15625 . doi: 10.1038/s41598-020-70542-9 . OpenUrl CrossRef 36. ↵ Yan K , Wang X , Lu L , Summers RM . DeepLesion: automated mining of large-scale lesion annotations and universal lesion detection with deep learning . J Med Imaging Bellingham Wash . 2018 ; 5 ( 3 ): 036501 . doi: 10.1117/1.JMI.5.3.036501 . OpenUrl CrossRef 37. ↵ Gouthamchand V , Fonseca LA , Hoebers FJ , et al. Performance of Handcrafted Radiomics versus Deep Learning for Prognosticating Head and Neck Squamous Cell Carcinoma – A Systematic Review with Critical Appraisal of Quantitative Imaging Studies . medRxiv ; 2024 . p. 2024.10.22.24315007. doi: 10.1101/2024.10.22.24315007 . OpenUrl Abstract / FREE Full Text 38. ↵ Leger S , Zwanenburg A , Leger K , et al. Comprehensive Analysis of Tumour Sub-Volumes for Radiomic Risk Modelling in Locally Advanced HNSCC . Cancers . 2020 ; 12 ( 10 ): 3047 . doi: 10.3390/cancers12103047 . OpenUrl CrossRef PubMed 39. ↵ van der Kamp MF , Muntinghe FOW , Iepsma RS , et al. Predictors for distant metastasis in head and neck cancer, with emphasis on age . Eur Arch Otorhinolaryngol . 2021 ; 278 ( 1 ): 181 – 190 . doi: 10.1007/s00405-020-06118-0 . OpenUrl CrossRef PubMed 40. ↵ Vaidya A , Chen RJ , Williamson DFK , et al. Demographic bias in misdiagnosis by computational pathology models . Nat Med. Nature Publishing Group ; 2024 ; 30 ( 4 ): 1174 – 1190 . doi: 10.1038/s41591-024-02885-z . OpenUrl CrossRef 41. ↵ Blankemeier L , Cohen JP , Kumar A , et al. Merlin: A Vision Language Foundation Model for 3D Computed Tomography . arXiv; 2024 . doi: 10.48550/arXiv.2406.06512 . OpenUrl CrossRef 42. ↵ Hamamci IE , Er S , Almas F , et al. A foundation model utilizing chest CT volumes and radiology reports for supervised-level zero-shot detection of abnormalities. arXiv ; 2024 . doi: 10.48550/arXiv.2403.17834 . OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted January 25, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following End-to-end prediction of clinical outcomes in head and neck squamous cell carcinoma with foundation model-based multiple instance learning Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share End-to-end prediction of clinical outcomes in head and neck squamous cell carcinoma with foundation model-based multiple instance learning Asier Rabasco Meneghetti , Marta Ligero Hernández , Jens-Peter Kuehn , Steffen Löck , Zunamys Itzel Carrero , Raquel Perez-Lopez , Keno Bressem , Titus K. Brinker , Alexander T. Pearson , Daniel Truhn , Sven Nebelung , Jakob Nikolas Kather medRxiv 2025.01.22.25320517; doi: https://doi.org/10.1101/2025.01.22.25320517 Share This Article: Copy Citation Tools End-to-end prediction of clinical outcomes in head and neck squamous cell carcinoma with foundation model-based multiple instance learning Asier Rabasco Meneghetti , Marta Ligero Hernández , Jens-Peter Kuehn , Steffen Löck , Zunamys Itzel Carrero , Raquel Perez-Lopez , Keno Bressem , Titus K. Brinker , Alexander T. Pearson , Daniel Truhn , Sven Nebelung , Jakob Nikolas Kather medRxiv 2025.01.22.25320517; doi: https://doi.org/10.1101/2025.01.22.25320517 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Oncology Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (297) Cardiovascular Medicine (4421) Dentistry and Oral Medicine (443) Dermatology (382) Emergency Medicine (606) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1507) Epidemiology (15212) Forensic Medicine (30) Gastroenterology (1121) Genetic and Genomic Medicine (6581) Geriatric Medicine (667) Health Economics (996) Health Informatics (4520) Health Policy (1366) Health Systems and Quality Improvement (1611) Hematology (539) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15906) Intensive Care and Critical Care Medicine (1103) Medical Education (620) Medical Ethics (144) Nephrology (667) Neurology (6580) Nursing (345) Nutrition (998) Obstetrics and Gynecology (1141) Occupational and Environmental Health (956) Oncology (3324) Ophthalmology (970) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (663) Pediatrics (1689) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5433) Public and Global Health (9212) Radiology and Imaging (2193) Rehabilitation Medicine and Physical Therapy (1368) Respiratory Medicine (1194) Rheumatology (593) Sexual and Reproductive Health (709) Sports Medicine (529) Surgery (709) Toxicology (99) Transplantation (288) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9ff56eb5af3d593a',t:'MTc3OTM4NTkyMA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-21T05:10:58.409756+00:00
License: CC-BY-4.0