Full text
49,951 characters
· extracted from
preprint-html
· click to expand
Predicting Cardiopulmonary Exercise Testing Performance in Patients Undergoing Transthoracic Echocardiography - An AI Based, Multimodal Model | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Predicting Cardiopulmonary Exercise Testing Performance in Patients Undergoing Transthoracic Echocardiography - An AI Based, Multimodal Model Shudhanshu Alishetti , Weishen Pan , Ashley N. Beecy , Zhenzhen Liu , Albert Gong , Zhe Huang , Kevin J. Clerkin , Rochelle L. Goldsmith , David T. Majure , Chris Kelsey , David vanMaanan , Jeffrey Ruhl , Naomi Tesfuzigta , Erica Lancet , Deepa Kumaraiah , Gabriel Sayer , Deborah Estrin , Kilian Weinberger , Volodymyr Kuleshov , Fei Wang , Nir Uriel doi: https://doi.org/10.1101/2025.07.05.25330921 Shudhanshu Alishetti a Department of Cardiology, NewYork Presbyterian – Brooklyn Methodist Hospital , Brooklyn, New York, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: sha9160{at}nyp.org Weishen Pan b Department of Population Health Sciences, Weill Cornell Medicine, Cornell University , New York, New York, USA PHD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ashley N. Beecy c Division of Cardiology, Department of Medicine, Weill Cornell Medicine , New York, New York, USA d Information Technology Data Science, NewYork Presbyterian Hospital , New York, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Zhenzhen Liu e Department of Computer Science, Cornell University , Ithaca, New York, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Albert Gong e Department of Computer Science, Cornell University , Ithaca, New York, USA BS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Zhe Huang b Department of Population Health Sciences, Weill Cornell Medicine, Cornell University , New York, New York, USA PHD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kevin J. Clerkin f Seymour, Paul, and Gloria Milstein Division of Cardiology, Department of Medicine, Columbia University Irving Medical Center/NewYork Presbyterian Hospital , New York, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Rochelle L. Goldsmith f Seymour, Paul, and Gloria Milstein Division of Cardiology, Department of Medicine, Columbia University Irving Medical Center/NewYork Presbyterian Hospital , New York, USA PHD Find this author on Google Scholar Find this author on PubMed Search for this author on this site David T. Majure c Division of Cardiology, Department of Medicine, Weill Cornell Medicine , New York, New York, USA MD, MPH Find this author on Google Scholar Find this author on PubMed Search for this author on this site Chris Kelsey d Information Technology Data Science, NewYork Presbyterian Hospital , New York, USA BA Find this author on Google Scholar Find this author on PubMed Search for this author on this site David vanMaanan d Information Technology Data Science, NewYork Presbyterian Hospital , New York, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jeffrey Ruhl d Information Technology Data Science, NewYork Presbyterian Hospital , New York, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Naomi Tesfuzigta g Cornell Tech , New York, New York, USA MPH Find this author on Google Scholar Find this author on PubMed Search for this author on this site Erica Lancet h NewYork Presbyterian Hospital , New York, New York, USA MPA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Deepa Kumaraiah d Information Technology Data Science, NewYork Presbyterian Hospital , New York, USA f Seymour, Paul, and Gloria Milstein Division of Cardiology, Department of Medicine, Columbia University Irving Medical Center/NewYork Presbyterian Hospital , New York, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Gabriel Sayer f Seymour, Paul, and Gloria Milstein Division of Cardiology, Department of Medicine, Columbia University Irving Medical Center/NewYork Presbyterian Hospital , New York, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Deborah Estrin b Department of Population Health Sciences, Weill Cornell Medicine, Cornell University , New York, New York, USA e Department of Computer Science, Cornell University , Ithaca, New York, USA g Cornell Tech , New York, New York, USA PHD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kilian Weinberger e Department of Computer Science, Cornell University , Ithaca, New York, USA PHD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Volodymyr Kuleshov g Cornell Tech , New York, New York, USA PHD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Fei Wang b Department of Population Health Sciences, Weill Cornell Medicine, Cornell University , New York, New York, USA PHD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nir Uriel f Seymour, Paul, and Gloria Milstein Division of Cardiology, Department of Medicine, Columbia University Irving Medical Center/NewYork Presbyterian Hospital , New York, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Background and Aims Transthoracic echocardiography (TTE) is a widely available tool for diagnosing and managing heart failure but has limited predictive value for survival. Cardiopulmonary exercise test (CPET) performance strongly correlates with survival in heart failure patients but is less accessible. We sought to develop an artificial intelligence (AI) algorithm using TTE and electronic medical records to predict CPET peak oxygen consumption (peak VO 2 ) ≤ 14 mL/kg/min. Methods An AI model was trained to predict peak VO 2 ≤ 14 mL/kg/min from TTE images, structured TTE reports, demographics, medications, labs, and vitals. The training set included patients with a TTE within 6 months of a CPET. Performance was retrospectively tested in a held-out group from the development cohort and an external validation cohort. Results 1,127 CPET studies paired with concomitant TTE were identified. The best performance was achieved by using all components (TTE images, all structured clinical data). The model performed well at predicting a peak VO 2 ≤ 14 mL/kg/min, with an AUROC of 0.84 (development cohort) and 0.80 (external validation cohort). It performed consistently well using higher (≤ 18 mL/kg/min) and lower (≤ 12 mL/kg/min) cut-offs. Conclusions This multimodal AI model effectively categorized patients into low and high risk predicted peak VO 2 , demonstrating the potential to identify previously unrecognized patients in need of advanced heart failure therapies where CPET is not available. Introduction Heart failure (HF) impacts nearly 7 million in the United States, with that number expected to increase to more than 10 million by 2040( 1 ). HF is a progressive, fatal condition that is associated with a median loss of 15 years of life. It has the potential to impact all, agnostic to age, ethnicity, or location; however, some groups bear a greater burden. Patients in rural areas demonstrate higher HF mortality for both younger and older age groups, with mortality rates rising steadily since 2012. Despite improvements in medical therapy, HF related mortality is on the rise. One potential reason is inadequate awareness and recognition of patients with advanced HF. This subset of patients, an estimated 200,000 Americans, have survival worse than most cancers( 1 , 2 ). Advanced therapies (heart transplantation and left ventricular assist devices) can alter that trajectory, improving survival and quality of life. Without advanced therapies, patients with advanced heart failure have a one-year survival of 25-50%( 3 , 4 ). Yet only about 6,000 patients receive advanced HF therapies annually; the other 194,000 do not. Echocardiography plays a significant role in the diagnosis and management of patients with HF. It is ubiquitously available and provides an assessment of cardiac structure and function. However, it has had limited value in predicting survival in HF. Secondary analyses of the Val-HeFT and CHARM studies( 5 , 6 ) demonstrated worse survival with decreasing EF. However, the SOLVD trial saw worse survival in patients with symptomatic HF compared with asymptomatic HF, and interestingly did not see a significant difference in EF between patients with asymptomatic HF and those with moderate HF symptoms( 7 , 8 ). More recently a study of almost 1,000 patients who met the European Society of Cardiology definition of advanced heart failure found that ejection fraction was not associated with hospitalization or mortality( 3 ). Thus, while an important test, it cannot faithfully identify patients who have progressed from symptomatic HF (Stage C) to advanced HF (Stage D) where advanced therapies are necessary. Functional capacity has demonstrated a strong association with outcomes in HF and identification of advanced HF. Across numerous studies, patients with New York Heart Association (NYHA) functional class I symptoms had a 1 year mortality around 5%, rising to 15% for those with NYHA class II or III symptoms, and 44% for NYHA class IV( 7 – 9 ). Cardiopulmonary exercise testing (CPET) allows for further delineation of functional capacity beyond the NYHA functional classes. A seminal study by Mancini et al. found that peak VO 2 , an objective measure of exercise capacity, predicted survival and differentiated between when patients needed heart transplantation and when transplantation could be safely deferred( 10 ). Performance on these tests also correlates with HF prognosis( 11 , 12 ). However, CPETs are far less frequently administered than TTE. This is in part because of a limited number of facilities that perform CPETs as they require an exercise testing laboratory equipped with a metabolic cart as well as a medical professional capable of interpreting the test. Artificial intelligence (AI) applied to echocardiography has demonstrated the ability to quantify ejection fraction( 13 – 15 ), quantitate cardiac structures( 16 ), assess mitral regurgitation( 17 ), detect various disease states, and predict outcomes( 18 – 20 ). While such models aim to obtain information that trained echocardiography readers can gather from a study, AI techniques may allow for the extraction of information from echocardiograms that cannot currently be obtained using usual techniques, such as exercise capacity. In this study we sought to develop and evaluate an AI algorithm utilizing multi-modal architecture incorporating echocardiography data and demographic data readily available in an electronic medical record system to predict peak VO 2 . Methods Ethics approval This study was approved by the institutional review boards at Weill Cornell Medicine and Columbia University Irving Medical Center. A waiver for informed consent was obtained. Study Cohort This study leveraged a database from different sites of NewYork Presbyterian/Columbia University Irving Medical Center (CUMC), NewYork Presbyterian/Weill Cornell Medical Center (WCM), NewYork Presbyterian/Brooklyn Methodist Hospital, and NewYork Presbyterian Queens Hospital. The database includes data from cardiopulmonary exercise tests (CPET), transthoracic echocardiograms (TTE), and clinical data from the electronic health record (EHR) outlined below. Considering the sample sizes of the sites, data from CUMC were used to construct the training samples, while the samples from all the other sites were used for external validation. Data Curation and Preprocessing The study cohort included adult (age ≥ 18 years) patients at CUMC who underwent a CPET with adequate effort (respiratory exchange ratio (RER) ≥ 1.05) within six months of a transthoracic echocardiogram (TTE). Development Cohort Samples Since 2017 3,872 cardiopulmonary exercise testing studies have been conducted for adult patients at CUMC within 6 months of a TTE. Of those, 1,000 met the specific inclusion criteria and were used to train the model. Among the patients excluded, 2,454 patients were excluded because TTE images and videos were not available, 32 were excluded because TTEs were in the pediatric format, and 12 were excluded because no available TTE videos from the standard parasternal long axis (PLAX), apical four chamber (A4C) view types were in the corresponding echo studies, and 374 were excluded because of a sub-maximal CPET (respiratory exchange ratio<1.05) (Supplemental Figure 1). External Validation Samples An external validation set of 127 CPET studies which met inclusion criteria was compiled consisting of patients who had cardiopulmonary exercise testing completed at WCM, NewYork Presbyterian/Brooklyn Methodist Hospital, and NewYork Presbyterian Queens Hospital (Supplemental Figure 1). To construct the input, we first preprocessed the data by extracting the structured features from the electronic health records (EHRs) and identifying the types of videos and images from an TTE as follows: Structured Feature Extraction and Preprocessing Different categories of variables were considered to construct the structured feature vector to predict peak VO 2 , including demographics, concomitant medications, laboratory tests, vital signs and structural TTE measurement and findings. These variables were determined by clinical experts. The full list of the variables is presented in the Supplementary Table 1. And the preprocess procedures for different kinds of variables are illustrated as follows: Demographic variables Demographic variables included age at the CPET date, sex, and race. Concomitant medications Medications were grouped into pharmaceutical subclasses based on the anatomical therapeutic chemical (ATC) codes. Each medication feature denotes whether the medications from its respective group were used in the observation window. Laboratory tests and vital signs Laboratory tests and vital signs were identified from structured data using Logical Observations Identifier Names and Codes (LOINC) within the observation period. For each test, the value closest to the CPET date was utilized in the presence of multiple measurements. TTE measurement and findings The features of measurement and findings were extracted from the report of echo study closest to the target CPET. After constructing the features, the missing values were imputed as: ( 1 ) mode for categorical and binary features; and ( 2 ) mean for continuous features. The mode and mean values for imputation were calculated based on the training set of the development data (see Model Training, Development, and Validation for data splitting). In order to eliminate the effects of value magnitude, all variables were scaled according to the z-score. TTE Video/Image Identification and Preprocessing For TTE data, we classified the subtypes of TTE videos without color Doppler, M-mode images, and color Doppler images, respectively. The preprocess procedures for different kinds of TTE imaging data are illustrated as follows. TTE videos without color Doppler Specifically, we utilized a convolutional neural network (CNN) to identify the TTE videos as one of the standard echocardiographic views. In particular, we utilized the pre-trained model in the previous study to classify the TTE videos, which is publicly available and widely used( 13 ). With the pre-trained image-based classifier, we first fed each frame sub-sampled from a video to the classifier and averaged the resulting probability distributions to obtain the prediction result of the video. Finally, we selected the TTE videos with the standard parasternal long axis (PLAX), apical four chamber (A4C) views. After identifying and selecting the TTE videos, we further preprocessed the selected videos by cropping and masking them to remove the text and all other information outside the scanning sector. The resulting videos were further subsampled by one in every two frames and resized to 112 × 112-pixel video. M-mode images We manually labeled and selected the images based on the focused regions including the aortic valve (AO), mitral valve (MV) and left ventricle (LV). The selected M-mode images were further cropped to only keep the waveform region of which the coordinates were annotated in the Digital Imaging and Communications in Medicine (DICOM) files. The resulting images were resized to 224 × 224-pixel images. Color Doppler images We first divided them into tissue Doppler images, Doppler images with pulsed wave (PW) and Doppler Images with continuous wave (CW) based on the tag stored in the DICOM files. Then we trained CNNs to identify subtypes of these color Doppler images based on the focused regions and measurements with the labeled data from the external WCM cohort with available annotations and apply it to the data from the CUMC cohort. The selected Doppler images were further cropped to only keep the waveform region and remove the artifacts in the waveform region. The resulting images were resized to 224 × 224-pixel images. Supplementary Table 2 lists the types of TTE videos and images used in our model. Model Structure After extracting the structured features and identifying the TTE videos and images, we fed each into two separate branches to calculate the respective prediction scores. Finally, we assembled the prediction scores obtained from the structured features and TTE data to get the final prediction, as illustrated in Figure 1 . Download figure Open in new tab Figure 1. The workflow of the method. In the branch with the structured features, we tested 4 popular models including the logistic regression, multilayer perceptron random forest, and extreme gradient boosting models. The branch of the model with the TTE data is illustrated in Figure 1(b) . Since multiple TTE videos and images existed in each TTE study with arbitrary numbers from each subtype, we first fed the videos/images into category-specific encoders based on their categories (TTE videos without color Doppler, m-mode images, or color Doppler images) to extract the features. Then the extracted features were input into subtype-specific regressors based on their subtypes (e.g., from A4C or PLAX for TTE videos without color Doppler) to obtain the predicted peak VO 2 . We repeated the above procedure for all identified videos and images from the same study and then averaged the obtained predicted values as the final prediction for the outcome. In particular, we implemented the video encoder as a 3D residual convolutional neural network ( 21 ), which is the structure used in Echonet ( 14 ). And we implemented the image encoder for m-mode and Doppler images as a 2D residual convolutional neural network. Video Encoder Pretraining We first pretrained the video encoder on the publicly available Echonet-Dynamic dataset, and then fine-tuned on our CPET dataset. The Echonet-Dynamic dataset consists of 10,030 grayscale A4C videos with ejection fraction labels. For the pretraining, we followed the standard training procedure of Echonet and used an 18-layer (2+1)D ResNet ( 22 ) for ejection fraction prediction. Model Training, Development, and Validation We performed a 64%-16%-20% train-validation-test split, divided by patients, within the development cohort. To train the proposed model, we first generated video-label/image-label pairs by mapping the peak VO 2 label to each identified video and image from the linked TTE study. Then we trained the model by minimizing the mean square error on the training video-label/image-label pairs. During testing, we applied the trained model to each video/image from the same TTE study and then ensembled their results together. The training subset was used to train the model, with the hyperparameters tuned with the validation subset. Then the model was evaluated on the test subset of the development cohort, as well as the external validation cohort. We conducted the experiments with multiple randomly train-validation-test splits and reported the averaged performance. We evaluated the performance of the model to predict a peak VO 2 using a cut-off of ≤14 ml/kg/min. Additionally, we evaluated the model’s performance with the following tasks: ( 1 ) binary classification of peak VO 2 using different cut-offs (≤12 mL/kg/min and ≤18 mL/kg/min); ( 2 ) directly regress the peak VO 2 value. Results Data Curation and Preprocessing The study cohort included 1,127 total CPET studies in the development (n=1,000) and external validation cohorts (n=127). The characteristics of the two cohorts are shown in Table 1 . The two populations were similar with respect to age (mean 56.8±15.3 years vs. 56.0±15.8 years) and sex (35.8% female vs 35.4% female). The racial composition differed between the two cohorts with a greater proportion of White (47.8% vs. 40.2%) and other/declined (36.2% vs. 22.0%) patients and fewer Asian (2.5% vs. 18.1%) and Black patients (13.5% vs. 19.7%) in the development cohort. Baseline echocardiographic parameters were similar between the groups with EF (37.2%±16.7 vs. 39.9%±17.0), LV end diastolic diameter, and wall thickness having no clinically meaningful differences. Lastly the baseline peak VO 2 measurements were higher in the development cohort than the external validation cohort (20.1±7.7 mL/kg/min vs. 18.1±8.6 mL/kg/min). View this table: View inline View popup Table 1. Characteristics of the Development (Columbia) and External Validation (NYP-Brooklyn Methodist, NYP Queens, and Cornell) cohorts. The model’s efficacy was assessed on the held-out group from the development cohort and the external validation cohort. The performance of the model was assessed utilizing various components of the echocardiogram and clinical information. The first iteration involved the TTE images alone, specifically utilizing videos in the parasternal long axis and apical 4 chamber views, M-mode images, and pulsed wave and continuous wave Doppler images. Then the model was assessed utilizing the structured TTE report (the format of which differed at each site) that included qualitative and quantitative assessments of structure and function, with some demographic information including age, sex, height, weight, and body surface area. The final group (Structured [All]) included all the aforementioned information in addition to concomitant medications, laboratory values, vital signs, and demographics. We assessed combinations of the images and reports as well. We assessed the AI model’s ability to predict peak VO 2 ≤ 14 mL/kg/min using the data from the echocardiogram ( Figure 2 ). In the development held-out cohort, the model performed well with the area under the receiver operating characteristic curve (AUROC) values ranging from 0.79-0.84. Saliency mapping of PLAX and A4C TTE views indicated greater value in the regions of the left ventricular outflow tract, the medial mitral valve annulus, and the base of the left ventricle ( Figure 3 ). The most influential features involved age, congestion, and inflammation ( Figure 4 ). The best performance was with the inputs of echocardiogram images and the full structured data set (AUROC 0.84, Sensitivity 69.9% given 80% specificity, Specificity 71.5% given 80% sensitivity). The results were consistent in the external validation cohort, with an AUROC of 0.81 (Sensitivity 70.4% given 80% specificity, Specificity 68.2% given 80% sensitivity). There was incremental improvement in the classification metrics as the amount of input increased, with the best regression performance when both the TTE images and the complete structured report were utilized. When the threshold classification value for peak VO2 was adjusted, the model performed consistently well. Utilizing a peak VO2≤12 mL/kg/min (the guideline recommended cutoff for heart transplantation listing), the model performed well for the development held-out cohort (AUROC 0.82) and for the external validation cohort (AUROC 0.74, Table 2 ). When a more lenient cut-off of VO2≤18 mL/kg/min was applied, the model again demonstrated efficacy in the Columbia held-out (AUROC 0.86) and the external validation cohort (AUROC 0.81). The effectiveness of pretraining the video encoder with Echonet-Dynamic data is shown in Supplemental Table 3. Download figure Open in new tab Figure 2. Performance evaluation of the binary classification task for peak VO 2 (≤14 mL/kg/min). (a): Quantitative metrics in the Columbia held-out test data and Cornell data (external validation). Results are averaged across the 5 predefined splits in the Columbia cohort with standard deviations in parentheses. Structured (Echo Report) includes all the structured features in an echo report (echo measurements & findings and some demographics), Structured (All) includes all structured features (echo measurements & findings, vitals, labs, medications, and demographic). AUROC, the area under the receiver operating characteristic curve. (b) and (c): The receiver operating characteristic (ROC) curves for (b) Columbia held-out test and (c) WCM & NYP external validation in the first split. Download figure Open in new tab Figure 3. Model interpretability on echo imaging data. Sample visualizations of Grad-CAM saliency maps overlaid on echo videos for two patients: (a) peak VO 2 > 14; (b) peak VO 2 < 14. Download figure Open in new tab Figure 4. Model interpretability on structured data. The global Shapley value of the top-20 features on the prediction of the model. The features are ranked based on their global Shapley values shown on the x-axis. View this table: View inline View popup Download powerpoint Table 2. Results of different cutoffs for peak VO 2 (≤ 12, ≤ 18 mL/kg/min) in the Columbia held-out test data and external validation data. Results are averaged across the 5 predefined splits in the Columbia cohort with standard deviations in parentheses. Structured (Echo Report) includes the structured features in echo report (echo measurements & findings and some demographics), Structured (All) includes all structured features (echo measurements & findings, vitals, labs, medications, and demographic). AUROC, the area under the receiver operating characteristic curve. Given the baseline differences between the groups, we assessed the AI model’s consistency stratified by a number of patient baseline demographics. As shown in Table 3 , there was consistency between males and females with the AUROC for a peak VO 2 cutoff of 12 mL/kg/min (0.81 vs. 0.90), 14 mL/kg/min (0.83 vs. 0.89), and 18 mL/kg/min (0.86 vs. 0.89). It was similar for the external validation cohort. There again was consistency for non-white vs. white patients in both the development held-out and external validation cohorts ( Table 3 ). There were differences, however, across age groups. Among patients below the age of 60 the model had good fit and performed well with each peak VO 2 cutoff for both the internal and external cohorts. The AUROC remained reasonable in the internal validation cohort (>0.72) but declined in the external validation cohort ( Table 3 ). The results further indicate that patients with a shorter interval between cardiopulmonary exercise testing (CPET) and transthoracic echocardiography (TTE) studies—specifically, less than one month—achieved higher AUROC values. This pattern was consistently observed across both internal and external cohorts, highlighting the advantage of utilizing temporally proximate data for predictive modeling in clinical settings. Assessing the cohort stratified by renal function, similar to older age, the model did not perform as well for those with chronic kidney disease ( Table 3 ). View this table: View inline View popup Download powerpoint Table 3. Results with Echo Imaging + Structured (All) under different subgroups for Columbia held-out test and WCM & NYP external validation. Results are averaged across the 5 predefined splits in the Columbia cohort. Discussion Artificial intelligence (in some form) is ubiquitous in our daily lives, however widespread excitement (and disquietude) has abounded in the ChatGPT era. Within medicine, there is similar excitement (and fear) about the role AI can play in patient care moving forward. In this study we sought to leverage echocardiography performed as part of standard clinical care and generate clinically actionable information that previously required additional specialized testing (peak VO 2 ). Our principal findings were: 1) A novel AI algorithm demonstrated the ability to predict a patient’s peak VO 2 across the spectrum of functional impairment (peak VO 2 ≤12 mL/kg/min or peak VO 2 ≤18 mL/kg/min). 2) The model performed consistently in diverse internal validation and external validation cohorts. 3) The model performance was consistent across gender and race but demonstrated decreased performance in an older population. The promise of AI to improve medical care is great, and specifically within the field of cardiology tools have been developed to help augment echocardiography. Studies have demonstrated the ability of AI models to accurately quantify ejection fraction, measure chamber size, detect valvular disease, and quantify regurgitation. The automation of these tasks holds the potential to improve the efficiency of the performing sonographer or the interpreting cardiologist, but do not generate novel information. We were able to accomplish that, developing a model that can predict a high risk peak VO 2 without a concomitant cardiopulmonary exercise test. Utilizing a cut-off of 14 mL/kg/min, guided by the seminal paper by Mancini et al., the model performed well in both internal (AUROC 0.84) and external validation cohorts (AUROC 0.81). We chose 14 mL/kg/min for the screening cut-off as it identifies patients who have or are nearing advanced heart failure, though we felt it was also important to assess the model’s versatility. Utilizing the International Society for Heart and Lung Transplantation guideline recommended cut-off of 12 mL/kg/min (for patients on beta-blockers) for heart transplant listing, the model continued to perform well to identify the sickest patients in internal (AUROC 0.82) and external (0.74) validation cohorts. We also wanted to ensure the model continued to demonstrate efficacy for those with more mild functional limitations, and that was shown with the performance for a peak VO 2 ≤18 mL/kg/min. The prevalence of heart failure continues to rise with 6 million (1.7% of the United States population) impacted. There are only about 1,400 advanced heart failure trained physicians in the United States and >40% of United States hospitals with cardiology fellowship programs do not have an advanced heart failure trained physician. Finding who among those 6 million patients should meet one of the 1,400 requires assiduous assessment. Cardiopulmonary exercise testing is a specialized test that can help identify patients who benefit from advanced therapies (e.g. heart transplantation or left ventricular assist device). However, with specialization comes scarcity and CPET is not nearly as ubiquitous as echocardiography. The model developed in this study has the potential to bridge this divide, and with greater precision identify patients who would benefit from referral for more specialized care. We hope that this tool will help democratize care and identify the 97% of patients with advanced heart failure who don’t receive advanced therapies to improve patient outcomes. The model was not perfect, though, and demonstrated decreased performance in older patients (age > 60 years) and patients with chronic kidney disease. Cardiopulmonary exercise testing is a valid assessment tool in older adults with peak VO 2 being associated with functional capacity and mortality in this population. However, a reduced peak VO 2 does not always represent a cardiac deficiency; reduced peak VO 2 may also be from a pulmonary limitation (perfusion, diffusion, ventilation) or peripheral oxygen extraction limitation (skeletal muscle function, capillary density, mitochondrial function, anemia). It is possible that older patients, like patients with chronic kidney disease, have a greater burden of comorbidities that may impact the peak VO 2 and not be detected on the echocardiogram. Of note, when the model was trained to predict [the percent of predicted VO2 34], it did not perform as well as predicting peak VO2 (Supplemental Table 4). The reduced efficacy in predicting percent predicted VO2 may be because the model already incorporates age and weight into its prediction. Similarly, VE/VCO2 or ventilatory efficiency is not solely attributable to the heart, and can reflect pulmonary hypertension, chronic obstructive pulmonary disease, or other conditions affecting the lungs. Further prospective validation of this model will be important to further assess this observation. The study has several limitations. First is that the study was a retrospective analysis of patients referred for cardiopulmonary exercise testing, which may introduce confounding by indication; even as the model performed consistently across the spectrum of peak VO 2 . Next, the model was generated utilizing patients from four academic hospitals all in the New York area, potentially limiting generalizability. Additionally, the echocardiograms and the cardiopulmonary exercise tests were contemporaneous but not simultaneous, and it is possible that clinical changes may have occurred between the studies. Finally, we utilized the absolute peak VO 2 value, while percent predicted peak VO 2 may also be used for younger patients (<50 years). Nonetheless, the model performed quite well in this age range. Prospective validation will be required. CONCLUSIONS This multimodal AI model effectively categorized patients into low and high risk predicted peak VO 2 groups using TTE and clinical data. It demonstrates the potential to identify previously unrecognized patients in need of advanced heart failure therapies where CPET is not available. Data Availability The data underlying this article cannot be shared publicly due to restrictions of Weill Cornell Medicine and Columbia University Irving Medical Center on patient data. Footnotes The institution where the work was performed: NewYork-Presbyterian/Columbia University Irving Medical Center, NewYork-Presbyterian/Weill Cornell Medical Center, NewYork-Presbyterian/Brooklyn Methodist Hospital NewYork-Presbyterian/Queens Hospital ABBREVIATIONS LIST A4C apical four chamber AO aortic valve CPET cardiopulmonary exercise testing CW continuous wave LV left ventricle MV mitral valve PW pulse wave RER respiratory exchange ratio TTE transthoracic echocardiography VO 2 oxygen consumption REFERENCES 1. ↵ Bozkurt B , Ahmad T , Alexander K et al. HF STATS 2024: Heart Failure Epidemiology and Outcomes Statistics An Updated 2024 Report from the Heart Failure Society of America . J Card Fail 2024 . 2. ↵ Mamas MA , Sperrin M , Watson MC et al. Do patients have worse outcomes in heart failure than in cancer? A primary care-based cohort study with 10-year follow-up in Scotland . Eur J Heart Fail 2017 ; 19 : 1095 – 1104 . OpenUrl CrossRef PubMed 3. ↵ Dunlay SM , Roger VL , Killian JM et al. Advanced Heart Failure Epidemiology and Outcomes: A Population-Based Study . JACC Heart Fail 2021 ; 9 : 722 – 732 . OpenUrl PubMed 4. ↵ Rose EA , Gelijns AC , Moskowitz AJ et al. Long-term use of a left ventricular assist device for end-stage heart failure . N Engl J Med 2001 ; 345 : 1435 – 43 . OpenUrl CrossRef PubMed Web of Science 5. ↵ Wong M , Staszewsky L , Latini R et al. Severity of left ventricular remodeling defines outcomes and response to therapy in heart failure: Valsartan heart failure trial (Val-HeFT) echocardiographic data . J Am Coll Cardiol 2004 ; 43 : 2022 – 7 . OpenUrl FREE Full Text 6. ↵ Solomon SD , Anavekar N , Skali H et al. Influence of ejection fraction on cardiovascular outcomes in a broad spectrum of heart failure patients . Circulation 2005 ; 112 : 3738 – 44 . OpenUrl Abstract / FREE Full Text 7. ↵ Investigators S , Yusuf S , Pitt B , Davis CE , Hood WB , Cohn JN . Effect of enalapril on survival in patients with reduced left ventricular ejection fractions and congestive heart failure . N Engl J Med 1991 ; 325 : 293 – 302 . OpenUrl CrossRef PubMed Web of Science 8. ↵ Investigators S , Yusuf S , Pitt B , Davis CE , Hood WB , Jr. , Cohn JN . Effect of enalapril on mortality and the development of heart failure in asymptomatic patients with reduced left ventricular ejection fractions . N Engl J Med 1992 ; 327 : 685 – 91 . OpenUrl CrossRef PubMed Web of Science 9. ↵ Group CTS . Effects of enalapril on mortality in severe congestive heart failure. Results of the Cooperative North Scandinavian Enalapril Survival Study (CONSENSUS) . N Engl J Med 1987 ; 316 : 1429 – 35 . OpenUrl CrossRef PubMed Web of Science 10. ↵ Mancini DM , Eisen H , Kussmaul W , Mull R , Edmunds LH , Jr. , Wilson JR . Value of peak exercise oxygen consumption for optimal timing of cardiac transplantation in ambulatory patients with heart failure . Circulation 1991 ; 83 : 778 – 86 . OpenUrl Abstract / FREE Full Text 11. ↵ Roul G , Germain P , Bareiss P . Does the 6-minute walk test predict the prognosis in patients with NYHA class II or III chronic heart failure? Am Heart J 1998 ; 136 : 449 – 57 . OpenUrl CrossRef PubMed Web of Science 12. ↵ Stelken AM , Younis LT , Jennison SH et al. Prognostic value of cardiopulmonary exercise testing using percent achieved of predicted peak oxygen uptake for patients with ischemic and dilated cardiomyopathy . J Am Coll Cardiol 1996 ; 27 : 345 – 52 . OpenUrl FREE Full Text 13. ↵ Zhang J , Gajjala S , Agrawal P et al. Fully Automated Echocardiogram Interpretation in Clinical Practice . Circulation 2018 ; 138 : 1623 – 1635 . OpenUrl CrossRef PubMed 14. ↵ Ouyang D , He B , Ghorbani A et al. Video-based AI for beat-to-beat assessment of cardiac function . Nature 2020 ; 580 : 252 – 256 . OpenUrl CrossRef PubMed 15. ↵ He B , Kwan AC , Cho JH et al. Blinded, randomized trial of sonographer versus AI cardiac function assessment . Nature 2023 ; 616 : 520 – 524 . OpenUrl PubMed 16. ↵ Duffy G , Cheng PP , Yuan N et al. High-Throughput Precision Phenotyping of Left Ventricular Hypertrophy With Cardiovascular Deep Learning . JAMA Cardiol 2022 ; 7 : 386 – 395 . OpenUrl PubMed 17. ↵ Long A , Haggerty CM , Finer J et al. Deep Learning for Echo Analysis, Tracking, and Evaluation of Mitral Regurgitation (DELINEATE-MR) . Circulation 2024 ; 150 : 911 – 922 . OpenUrl PubMed 18. ↵ Ulloa Cerna AE , Jing L , Good CW et al. Deep-learning-assisted analysis of echocardiographic videos improves predictions of all-cause mortality . Nat Biomed Eng 2021 ; 5 : 546 – 554 . OpenUrl PubMed 19. Lau ES , Di Achille P , Kopparapu K et al. Deep Learning-Enabled Assessment of Left Heart Structure and Function Predicts Cardiovascular Outcomes . J Am Coll Cardiol 2023 ; 82 : 1936 – 1948 . OpenUrl PubMed 20. ↵ Holste G , Oikonomou EK , Mortazavi BJ et al. Severe aortic stenosis detection by deep learning applied to echocardiography . Eur Heart J 2023 ; 44 : 4592 – 4604 . OpenUrl PubMed 21. ↵ He K , Zhang X , Ren S , Sun J . Deep Residual Learning for Image Recognition . 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) 2015 : 770 - 778 . 22. ↵ Tran D , Wang H , Torresani L , Ray J , LeCun Y , Paluri M . A Closer Look at Spatiotemporal Convolutions for Action Recognition . 2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition 2017 : 6450 - 6459 . 23. Bozkurt B , Ahmad T , Alexander KM et al. Heart Failure Epidemiology and Outcomes Statistics: A Report of the Heart Failure Society of America . J Card Fail 2023 ; 29 : 1412 – 1451 . OpenUrl CrossRef PubMed 24. Costanzo MR , Mills RM , Wynne J . Characteristics of “Stage D” heart failure: insights from the Acute Decompensated Heart Failure National Registry Longitudinal Module (ADHERE LM) . Am Heart J 2008 ; 155 : 339 – 47 . OpenUrl CrossRef PubMed 25. Yuzefpolskaya M , Schroeder SE , Houston BA et al. The Society of Thoracic Surgeons Intermacs 2022 Annual Report: Focus on the 2018 Heart Transplant Allocation System . Ann Thorac Surg 2023 ; 115 : 311 – 327 . OpenUrl CrossRef PubMed 26. Colvin MM , Smith JM , Ahn YS , et al. OPTN/SRTR 2021 Annual Data Report: Heart . Am J Transplant 2023 ; 23 : S300 – S378 . OpenUrl CrossRef PubMed 27. Choi DJ , Park JJ , Ali T , Lee S . Artificial intelligence for the diagnosis of heart failure . NPJ Digit Med 2020 ; 3 : 54 . 28. Baashar Y , Alkawsi G , Alhussian H et al. Effectiveness of Artificial Intelligence Models for Cardiovascular Disease Prediction: Network Meta-Analysis . Comput Intell Neurosci 2022 ; 2022 : 5849995 . OpenUrl PubMed 29. Bachtiger P , Petri CF , Scott FE , et al. Point-of-care screening for heart failure with reduced ejection fraction using artificial intelligence during ECG-enabled stethoscope examination in London, UK: a prospective, observational, multicentre study . Lancet Digit Health 2022 ; 4 : e117 - e125 . OpenUrl CrossRef 30. Shin S , Austin PC , Ross HJ et al. Machine learning vs. conventional statistical models for predicting heart failure readmission and mortality . ESC Heart Fail 2021 ; 8 : 106 – 115 . OpenUrl CrossRef PubMed 31. Spathis D , Perez-Pozuelo I , Gonzales TI et al. Longitudinal cardio-respiratory fitness prediction through wearables in free-living environments . NPJ Digit Med 2022 ; 5 : 176 . View the discussion thread. Back to top Previous Next Posted July 06, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Predicting Cardiopulmonary Exercise Testing Performance in Patients Undergoing Transthoracic Echocardiography - An AI Based, Multimodal Model Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Predicting Cardiopulmonary Exercise Testing Performance in Patients Undergoing Transthoracic Echocardiography - An AI Based, Multimodal Model Shudhanshu Alishetti , Weishen Pan , Ashley N. Beecy , Zhenzhen Liu , Albert Gong , Zhe Huang , Kevin J. Clerkin , Rochelle L. Goldsmith , David T. Majure , Chris Kelsey , David vanMaanan , Jeffrey Ruhl , Naomi Tesfuzigta , Erica Lancet , Deepa Kumaraiah , Gabriel Sayer , Deborah Estrin , Kilian Weinberger , Volodymyr Kuleshov , Fei Wang , Nir Uriel medRxiv 2025.07.05.25330921; doi: https://doi.org/10.1101/2025.07.05.25330921 Share This Article: Copy Citation Tools Predicting Cardiopulmonary Exercise Testing Performance in Patients Undergoing Transthoracic Echocardiography - An AI Based, Multimodal Model Shudhanshu Alishetti , Weishen Pan , Ashley N. Beecy , Zhenzhen Liu , Albert Gong , Zhe Huang , Kevin J. Clerkin , Rochelle L. Goldsmith , David T. Majure , Chris Kelsey , David vanMaanan , Jeffrey Ruhl , Naomi Tesfuzigta , Erica Lancet , Deepa Kumaraiah , Gabriel Sayer , Deborah Estrin , Kilian Weinberger , Volodymyr Kuleshov , Fei Wang , Nir Uriel medRxiv 2025.07.05.25330921; doi: https://doi.org/10.1101/2025.07.05.25330921 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Cardiovascular Medicine Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (299) Cardiovascular Medicine (4425) Dentistry and Oral Medicine (443) Dermatology (382) Emergency Medicine (607) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1507) Epidemiology (15221) Forensic Medicine (30) Gastroenterology (1123) Genetic and Genomic Medicine (6588) Geriatric Medicine (667) Health Economics (997) Health Informatics (4524) Health Policy (1368) Health Systems and Quality Improvement (1612) Hematology (540) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15910) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (145) Nephrology (667) Neurology (6588) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1143) Occupational and Environmental Health (956) Oncology (3331) Ophthalmology (970) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (663) Pediatrics (1690) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5440) Public and Global Health (9219) Radiology and Imaging (2195) Rehabilitation Medicine and Physical Therapy (1369) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (710) Sports Medicine (529) Surgery (710) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9ffbc0e06991aa64',t:'MTc3OTQ1MjIwMQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.