Leveraging a Vision-Language Model with Natural Text Supervision for MRI Retrieval, Captioning, Classification, and Visual Question Answering

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Large multimodal models are now extensively used worldwide, with the most powerful ones trained on massive, general-purpose datasets. Despite their rapid deployment, concerns persist regarding the quality and domain relevance of the training data, especially in radiology, medical research, and neuroscience. Additionally, healthcare data privacy is paramount when querying models trained on medical data, as is transparency regarding service hosting and data storage. So far, most deep learning algorithms in radiologic research are designed to perform a specific task (e.g., diagnostic classification) and cannot be prompted to perform multiple tasks using natural language. In this work, we introduce a framework based on vector retrieval and contrastive learning to efficiently learn visual brain MRI concepts via natural language supervision. We show how the method learns to identify factors that affect the brain in Alzheimer’s disease (AD) via joint embedding and natural language supervision. First, we pre-train separate text and image encoders using self-supervised learning, and jointly fine-tune these encoders to develop a shared embedding space. We train our model to perform multiple tasks, including MRI retrieval, MRI captioning, and MRI classification. We show its versatility by developing a retrieval and re-ranking mechanism along with a transformer decoder for visual question answering. Clinical Relevance By learning a cross-modal embedding of radiologic features and text, our approach can learn to perform diagnostic and prognostic assessments in AD research as well as to assist practicing clinicians. Integrating medical imaging with clinical descriptions and text prompts, we aim to provide a general, versatile tool for detecting radiologic features described by text, offering a new approach to radiologic research.
Full text 32,993 characters · extracted from preprint-html · click to expand
Leveraging a Vision-Language Model with Natural Text Supervision for MRI Retrieval, Captioning, Classification, and Visual Question Answering | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Leveraging a Vision-Language Model with Natural Text Supervision for MRI Retrieval, Captioning, Classification, and Visual Question Answering Nikhil J. Dhinagar , Sophia I. Thomopoulos , Paul M. Thompson doi: https://doi.org/10.1101/2025.02.15.638446 Nikhil J. Dhinagar 1 Imaging Genetics Center, Mark & Mary Stevens Neuroimaging & Informatics Institute, Keck School of Medicine, University of Southern California , Los Angeles, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: dhinagar{at}usc.edu Sophia I. Thomopoulos 1 Imaging Genetics Center, Mark & Mary Stevens Neuroimaging & Informatics Institute, Keck School of Medicine, University of Southern California , Los Angeles, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Paul M. Thompson 1 Imaging Genetics Center, Mark & Mary Stevens Neuroimaging & Informatics Institute, Keck School of Medicine, University of Southern California , Los Angeles, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Preview PDF Abstract Large multimodal models are now extensively used worldwide, with the most powerful ones trained on massive, general-purpose datasets. Despite their rapid deployment, concerns persist regarding the quality and domain relevance of the training data, especially in radiology, medical research, and neuroscience. Additionally, healthcare data privacy is paramount when querying models trained on medical data, as is transparency regarding service hosting and data storage. So far, most deep learning algorithms in radiologic research are designed to perform a specific task (e.g., diagnostic classification) and cannot be prompted to perform multiple tasks using natural language. In this work, we introduce a framework based on vector retrieval and contrastive learning to efficiently learn visual brain MRI concepts via natural language supervision. We show how the method learns to identify factors that affect the brain in Alzheimer’s disease (AD) via joint embedding and natural language supervision. First, we pre-train separate text and image encoders using self-supervised learning, and jointly fine-tune these encoders to develop a shared embedding space. We train our model to perform multiple tasks, including MRI retrieval, MRI captioning, and MRI classification. We show its versatility by developing a retrieval and re-ranking mechanism along with a transformer decoder for visual question answering. Clinical Relevance By learning a cross-modal embedding of radiologic features and text, our approach can learn to perform diagnostic and prognostic assessments in AD research as well as to assist practicing clinicians. Integrating medical imaging with clinical descriptions and text prompts, we aim to provide a general, versatile tool for detecting radiologic features described by text, offering a new approach to radiologic research. I. I ntroduction Alzheimer’s disease (AD) is a progressive neurodegenerative disease and the leading cause of dementia worldwide. In the United States, it is estimated that over 6 million individuals aged 65 and older are living with AD [ 1 ]. Globally, the burden of AD is substantial and is projected to increase dramatically in the coming decades as populations age [ 2 ]. Accurate diagnosis and prognosis of AD are critical for effective disease management. Current diagnostic criteria, as established by the National Institute on Aging (NIA), along with the Alzheimer’s Association, rely on a combination of clinical assessments and biomarker evidence. Imaging techniques such as positron emission tomography (PET) along with structural imaging methods such as magnetic resonance imaging (MRI), play a role in this diagnostic process [ 3 ]. In particular, T1-weighted brain MRI is widely used in both clinical practice and research settings as it maps and quantifies brain anatomy in detail, making it a valuable tool for assessing atrophy patterns associated with normal aging and AD [ 4 ]. Beyond imaging, multimodal data—including clinical evaluations, cognitive tests, and other biomarkers—are collected to capture the full complexity of AD pathology [ 5 ]. Integrating diverse data sources holds great promise for improving diagnostic accuracy and understanding factors that affect disease progression. Recent advances in artificial intelligence (AI), particularly in machine learning and deep learning, have further enabled researchers to analyze large, heterogeneous datasets. These AI-driven methods include convolutional neural networks, vision transformers [ 6 ] and latent diffusion models [ 7 ] that can be trained to identify subtle imaging biomarkers for disease classification or prognosis, as well as image segmentation and labeling [ 8 ]. Extensive benchmarking [ 9 ] [ 10 ] has led to powerful methods for disease detection. Even so, AI algorithms in radiology are still typically trained to perform a single task (e.g., disease classification) and cannot usually be trained to perform multiple tasks. Here we address this problem by training a vision-language model to perform multiple radiologic tasks (image retrieval, captioning, and classification) via natural language supervision. In computer vision, vision-language models have evolved from systems that classify images into a fixed set of categories to models that leverage natural language supervision for broader understanding. In the 2010s, models were typically trained to predict predetermined object categories. OpenAI’s CLIP model [ 11 ] shifted this paradigm by training a high-dimensional joint text-image embedding. The CLIP model was created by applying contrastive learning to 400 million image–text pairs, thereby achieving strong zero-shot classification without task-specific tuning. Subsequent developments include the Large Language and Vision Assistant (LLaVA) [ 12 ], which integrates a vision encoder with a large language model for end-to-end multimodal understanding. More recent advancements include Google’s PaliGemma2 [ 13 ], built on the Gemma2 language model, along with specialized models such as BioMedGPT [ 14 ] for biomedical tasks. In the medical imaging domain, MedBLIP [ 15 ] —a BLIP-2-based architecture—extends 2D models to 3D data, while NeuroBERT [ 16 ] demonstrates the potential of text-vision approaches for detecting abnormalities in brain MRI scans. Most large-scale vision-language models lack the capability to handle 3D medical imaging data natively. Using proprietary models raises concerns about training data quality, privacy issues and whether the training data is sufficiently relevant to be useful in specific domains such as radiology and clinical neuroscience. As well as imaging inputs, often demographic variables such as the patient’s age and sex [ 17 ] are important covariates for gauging AD pathology and disease-related atrophy, but they are often not readily integrated into diagnostic models that learn from images. In this paper, we propose a novel vision-language framework as shown in Figure 1 , designed to augment brain MRI analysis by integrating T1-weighted imaging data with text descriptions of subject-specific information, such as age, sex, and diagnosis. We evaluated our cross-modal model on multiple tasks - MRI retrieval, MRI captioning, and MRI classification. We illustrate the capability of our joint text-image vector embeddings with a retrieval and re-ranking mechanism using a text decoder for visual question answering. We conduct several ablation experiments to show the contributions of different components in our proposed framework. By training a deep learning model to learn connections between structural imaging and text-based data, we offer a comprehensive framework that could advance radiologic research, offering, in principle, a new approach to discover factors that affect disease progression. Download figure Open in new tab Figure 1. The Proposed Vision-Language Model. The left-hand side of the figure shows the possible user-side inputs (clinical text only, image only, image + clinical text, image + clinical text + user query). The right-hand side shows the response from the model (MRI retrieval, MRI captioning, MRI classification, visual question-answering). II. D ata In this study, we used the widely-used and publicly available ADNI (Alzheimer’s Disease Neuroimaging Initiative) brain MRI datasets for our analyses [ 18 ] ( Table 1 ). We pre-processed the 3D T1-weighted MRI scans using a standard set of pre-processing steps [ 19 ] including: nonparametric intensity normalization (N4 bias field correction) [ 20 ], ‘skull-stripping’, linear registration to a template with 9 degrees of freedom, and isotropic resampling of voxels to 2-mm resolution. The input dimension of the MRI spatially was 91×109×91. All images in the dataset was z-scored for standardization. View this table: View inline View popup Download powerpoint TABLE I. V olumetric T1- weighted B rain MRI dataset III. M ethods Stage 1: Uni-Modal Pre-training For our tasks, we want to be able to input text—either in the form of text prompts or queries, or as sentences describing the imaging data in text format—so their relationship can be learned. We tested several transformer-based text encoders to extract strong embeddings from the text inputs. This included BERT [ 21 ], ClinicalBioBERT [ 22 ], DeBERTa [ 23 ], BiomedBERT [ 24 ], and Sentence BERT [ 25 ]. We also created and evaluated different prompt structures to incorporate the covariates. Our final prompt format followed the style “78 year old male subject diagnosed [with] Alzheimer’s”. We selected the Sentence BERT as our main text encoder given its strong out-of-the-box performance on numerical and categorical concepts. Since “Alzheimer’s” was not already a part of the model’s vocabulary (the collection of words that the model can understand or generate), it was added to the tokenizer as a new token (in fact, we used the token “alzheimers” to standardize tokenization). We conducted domain-specific pretraining on the text encoder using masked language modeling [ 21 ] (MLM). We created a synthetic pre-training dataset of 25,000 text captions varying the age, sex and the disease descriptor. Here we masked 35% of the input text tokens and pre-trained the model to predict and thereby improve understanding of the numerical and categorical variables. Further, we used the powerful 3D DenseNet121 CNN [ 26 ] as our image encoder. This architecture has been proven [ 9 ] to be effective in extracting discriminative features for Alzheimer’s disease. The image encoder was pre-trained using contrastive learning and fully supervised learning with T1-weighted MRIs from ADNI [ 9 ]. The outputs of both the image and text encoder are passed through individual projection heads. Stage 2: Cross-Modal Contrastive Language-Image Fine-tuning Given a batch of N image-text pairs {( I i , T i )} N i=1 , the similarity matrix s i,j is defined as in equation (1), as the L2-normalized dot product between each image and text embedding: Here, f I and f T are the vision and text encoders; I i and T j are a given image and text pair. The symmetric cross-entropy loss ℒ is defined as in equation (2): Here, l i and l T are the image-to-text and text-to-image losses, and τ is the temperature parameter that controls the range of the logits – this parameter is directly optimized during training, as a log-parameterized multiplicative scalar. The pre-trained vision-language models were fine-tuned using different methods - full fine-tuning of all parameters; partial fine-tuning of all parameters after and including the last two layers of the encoders; locked image tuning [ 27 ] - freezing the image encoder backbone and fine-tuning everything else; or fine-tuning only the two projection heads. Stage 3: Applications The first application we trained the model to perform was MRI retrieval, i.e., given a text prompt, the framework retrieves the top- k image results corresponding to the text. Conversely, MRI captioning involves retrieving or generating the top- k text caption results given an input image. For MRI image classification, given an image and covariates (included as encoded text; see below), the model outputs a binary classification result that the patient has Alzheimer’s disease or is a healthy control. We also tested the effect of re-ranking on the retrieval results using a cross-encoder approach [ 28 ] [ 29 ]. We evaluated multiple state-of-the-art contemporary transformer-based decoder-only large language models (LLMs) for visual question answering including Google’s Gemma2 series [ 30 ], Meta’s Llama3 series [ 31 ], and Mistral’s 7B [ 32 ]. The Mistral 7B LLM was selected due to its strong ability to adhere to provided instructions. We used Meta AI’s optimized FAISS vector store [ 33 ] to create our database of vector embeddings for retrieval and re-ranking mechanisms for visual question answering. The retrieved results were re-ranked based on the corresponding covariates provided as an additional input. IV. E xperiments AND R esults A. Setup We performed a random search to select hyperparameter values, including the learning rate {1e-3 to 1e-6}, weight decay between 0.01 and 0.0001, batch size {4, 8, 16, 32, 64, 128, 256}, projection head dimension {256, 512, 1024}. The ADAM optimizer was used in all experiments. For the retrieval experiments, k was set to 10. A cosine learning rate scheduler with warmups was used for the image and text encoders, as well their respective projection heads. For our retrieval tasks (text-to-image and image-to-text) we used three standard evaluation metrics that are widely used in machine translation - specifically, bilingual evaluation understudy (BLEU), Recall-Oriented Understudy for Gisting Evaluation - Longest Common Subsequence (ROUGE-L)-- both to evaluate the quality of overlap between the retrieved result and the ground truth–as well BERTScore to quantify the semantic similarity between the two based on token embeddings. In addition, we used the receiver-operator characteristic curve-area under the curve (ROC-AUC), Accuracy for classification and Mean Absolute Error (MAE) for age prediction. The retrieval results and the VQA were calculated using over 100 unique retrievals. The classification results were calculated over 1,219 T1-weighted MRI scans. We performed a series of ablation experiments, including ablations within the text encoder, the text pre-training, the image encoder pre-training, the cross-modal fine-tuning method. These experiments are summarized in the following results sections per application. B. Application 1: MRI Retrieval We tested the fine-tuned cross-modal model for MRI retrieval using unique text captions sampled from the test set as the inputs. The results of these tests are presented in Table 2 as an average using the top-10 retrieved images. View this table: View inline View popup Download powerpoint TABLE II. MRI retrieval results (T ext-to -I mage ), where FT is fine-tune , LIT is locked image tuning , MLM is masked language modeling . C. Application 2: MRI Captioning We tested the fine-tuned, cross-modal model for MRI captioning, with unique images sampled from the test set as inputs. The results of these tests are summarized in Table 3 as an average using the top-10 retrieved text captions. View this table: View inline View popup Download powerpoint TABLE III. MRI C aptioning results (I mage-to -T ext ). D. Application 3: MRI classification We tested the fine-tuned cross-modal model for MRI classification with unique images sampled from the test set along with their covariates. Results of these tests are presented in Table 4 as an average using the top-10 retrieved text captions along with ablation runs. View this table: View inline View popup Download powerpoint TABLE IV. MRI C lassification results . E. Application 4: Visual Question Answering A sample VQA interaction with our model is shown in Table 5 . Here the model was given an image along with an instruction and query. Additionally, covariates are provided as encoded text to the model and used to improve the retrieval performance via re-ranking. The responses from the model are used to quantify its performance using average MAE (in years) for age prediction and average accuracy for disease classification, over the test set of T1-weighted MRIs. View this table: View inline View popup Download powerpoint TABLE V. V isual Q uestion A nswering results . View this table: View inline View popup Download powerpoint TABLE VI. S ample VQA interaction with the proposed model, given that the ground truth is: 78 year old male diagnosed with A lzheimer’s disease . V. D iscussion A. The pretraining and encoder architectures provide critical initialization for cross-modal retrieval tasks In our experiments, the text encoder and its pre-training corpus played a key role in cross-modal retrieval and classification performance, as seen in Tables 2 , 3 , and 4 . We observed that most ‘off-the-shelf’ text encoders were not sensitive to numerical and categorical concepts that are crucial to neuroimaging data. We also found that the cross-modal tasks were influenced by the image encoder’s pre-training. Specifically, while supervised pre-training improved downstream classification of the same task, contrastive pre-training provided a more balanced overall performance across all applications tested. B. Cross-modal fine-tuning methods can be more parameter-efficient but still maintain performance We evaluated different cross-modal fine-tuning methods - even though fully fine-tuning all layers usually yielded the best performance, the locked image tuning – which fine-tuned only the text backbone along with the image projection head – had competitive performance. This greatly reduces the total number of tunable parameters, and lends itself to low-resource settings. C. Strong vector embeddings lead to strong downstream results We created a vector database from our trained cross-modal model to facilitate visual question answering by extending the capabilities of an instruction-tuned transformer decoder model. The VQA performance on key benchmarking tasks such age prediction and disease detection were competitive with stand-alone regression and classification models. This shows that a single cross-modal model can handle multiple tasks without any additional fine-tuning. It is perhaps surprising that strong predictive models of age and disease are implicitly learned from the text-image embeddings, whereas usually classifiers are custom-trained using labeled training data, where the labels are typically a single parameter (age or diagnosis). Future work will include additional evaluation of our vision-language model on other datasets, modalities, and disorders. VI. C onclusion In this work, we presented a novel vector retrieval-driven vision-language framework that enhances brain image analysis by integrating T1-weighted imaging with complementary non-imaging clinical data. Although the approach is quite general and would be applicable to a variety of applications in radiology and medical imaging, we illustrated the approach on several tasks that arise in Alzheimer’s disease research. In the future, using an LLM interface, such tasks could be composed to perform virtual experiments, such as identifying brain abnormalities in a group of patients, or discovering imaging patterns associated with a medication or risk factor described in the associated text. Our approach leverages cross-modal retrieval techniques and fine-tuning of multimodal embeddings to enhance diagnostic assessments, providing a framework to capture subtle disease-related patterns that may be overlooked by single-modality analyses. Overall, this study offers a promising step toward more robust, data-driven approaches in radiologic research, with illustrative examples highlighting the role of cross-modal analysis in advancing our understanding of neurodegenerative diseases. Footnotes This work was supported by the U.S. National Institutes of Health, under NIH grant U01 AG068057 (‘AI4AD’). R eferences [1]. ↵ National Institute of Aging , “Alzheimer’s Disease Fact Sheet.” https://www.nia.nih.gov/health/alzheimers-and-dementia/alzheimers-disease-fact-sheet . [2]. ↵ E. Nichols et al. , “ Global, regional, and national burden of Alzheimer’s disease and other dementias, 1990–2016: a systematic analysis for the Global Burden of Disease Study 2016 ,” Lancet Neurol ., vol. 18 , no. 1 , pp. 88 – 106 , 2019 , doi: 10.1016/S1474-4422(18)30403-4 . OpenUrl CrossRef PubMed [3]. ↵ C. R. Jack et al. , “ Revised criteria for diagnosis and staging of Alzheimer’s disease: Alzheimer’s Association Workgroup ,” Alzheimer’s Dement ., vol. 20 , no. 8 , pp. 5143 – 5169 , 2024 , doi: 10.1002/alz.13859 . OpenUrl CrossRef PubMed [4]. ↵ G. B. Frisoni , N. C. Fox , C. R. Jack , P. Scheltens , and P. M. Thompson , “ The clinical use of structural MRI in Alzheimer disease ,” Nat. Rev. Neurol ., vol. 6 , no. 2 , pp. 67 – 77 , 2010 , doi: 10.1038/nrneurol.2009.215 . OpenUrl CrossRef PubMed Web of Science [5]. ↵ B. Dubois et al. , “ Preclinical Alzheimer’s disease: Definition, natural history, and diagnostic criteria ,” Alzheimers Dement ., vol. 12 , no. 3 , pp. 292 – 323 , 2019 , doi: 10.1016/j.jalz.2016.02.002.Preclinical . OpenUrl CrossRef [6]. ↵ N. J. Dhinagar , S. I. Thomopoulos , E. Laltoo , and P. M. Thompson , “ Efficiently Training Vision Transformers on Structural MRI Scans for Alzheimer’s Disease Detection ,” Proc. Annu. Int. Conf. IEEE Eng. Med. Biol. Soc. EMBS , 2023 , doi: 10.1109/EMBC40787.2023.10341190 . OpenUrl CrossRef [7]. ↵ N. J. Dhinagar , S. I. Thomopoulos , E. Laltoo , and P. M. Thompson , “ Counterfactual MRI Generation with Denoising Diffusion Models for Interpretable Alzheimer’s Disease Effect Detection ,” IEEE EMBC , 2024 , [Online]. Available: https://www.biorxiv.org/content/10.1101/2024.02.05.578983v1.abstract . [8]. ↵ M. J. Willemink , H. R. Roth , and V. Sandfort , “ Toward Foundational Deep Learning Models for Medical Imaging in the New Era of Transformer Networks ,” Radiol. Artif. Intell ., vol. 4 , no. 6 , 2022 . [9]. ↵ N. J. Dhinagar et al. , “ Video and synthetic MRI pre-training of 3D vision architectures for neuroimage analysis ,” SPIE Med. Imaging , 2024 , doi: 10.1117/12.3008837 . OpenUrl CrossRef [10]. ↵ B. Lu et al. , “ A practical Alzheimer’s disease classifier via brain imaging-based deep learning on 85,721 samples ,” J. Big Data , vol. 9 , no. 101 , 2022 . [11]. ↵ A. Radford et al. , “ Learning Transferable Visual Models From Natural Language Supervision ,” Proceedings of the 38th International Conference on Machine Learning (ICML) , 2021 . [12]. ↵ H. Liu , C. Li , Q. Wu , and Y. J. Lee , “Visual Instruction Tuning,” 2023 , no. NeurIPS , pp. 34892 – 34916 , [Online]. Available: http://arxiv.org/abs/2304.08485 . [13]. ↵ A. Steiner , A. S. Pinto , M. Tschannen , D. Keysers , X. Wang , and Y. Bitton , “ PaliGemma 2 : A Family of Versatile VLMs for Transfer ,” arXiv , 2024 . [14]. ↵ K. Zhang et al. , “ BiomedGPT: A Unified and Generalist Biomedical Generative Pre-trained Transformer for Vision, Language, and Multimodal Tasks ,” Nat. Med ., vol. 30 , pp. 3129 – 3141 , 2024 , [Online]. Available: http://arxiv.org/abs/2305.17100 . OpenUrl CrossRef PubMed [15]. ↵ Q. Chen , X. Hu , Z. Wang , and Y. Hong , “ MedBLIP: Bootstrapping Language-Image Pre-training from 3D Medical Images and Texts ,” arXiv , 2023 , [Online]. Available: http://arxiv.org/abs/2305.10799 . [16]. ↵ D. A. Wood , E. Guilhem , and J. H. Cole , “A self-supervised text-vision framework for automated brain abnormality detection,” 2024 . [Online]. Available: https://arxiv.org/abs/2405.02782 . [17]. ↵ A. S. Tang et al. , “ Leveraging electronic health records and knowledge networks for Alzheimer’s disease prediction and sex-specific biological insights ,” Nat. Aging , vol. 4 , no. 3 , pp. 379 – 395 , 2024 , doi: 10.1038/s43587-024-00573-8 . OpenUrl CrossRef PubMed [18]. ↵ M. W. Weiner and D. P. Veitch , “ Introduction to Special Issue Overview of ADNI ,” Alzheimers Dement , vol. 11 , no. 7 , pp. 730 – 733 , 2015 . OpenUrl CrossRef PubMed [19]. ↵ N. J. Dhinagar et al. , “ Evaluation of Transfer Learning Methods for Detecting Alzheimer’s Disease with Brain MRI ,” SIPAIM , pp. 504 – 513 , 2022 . [20]. ↵ N. J. Tustison , P. A. Cook , and J. C. Gee , “ N4ITK: Improved N3 Bias Correction ,” IEEE Trans Med Imaging ., vol. 29 , no. 6 , pp. 1310 – 1320 , 2010 , doi: 10.1109/TMI.2010.2046908.N4ITK . OpenUrl CrossRef PubMed Web of Science [21]. ↵ J. Devlin , M. W. Chang , K. Lee , and K. Toutanova , “ BERT: Pre-training of deep bidirectional transformers for language understanding ,” NAACL HLT 2019 - 2019 Conf. North Am. Chapter Assoc. Comput. Linguist. Hum. Lang. Technol. - Proc. Conf ., vol. 1 , no. Mlm , pp. 4171 – 4186 , 2019 . OpenUrl [22]. ↵ E. Alsentzer et al. , “Publicly Available Clinical BERT Embeddings,” 2019 , [Online]. Available: http://arxiv.org/abs/1904.03323 . [23]. ↵ P. He , X. Liu , J. Gao , and W. Chen , “ Deberta: Decoding-Enhanced Bert With Disentangled Attention ,” ICLR 2021 - 9th Int. Conf. Learn. Represent ., 2021 . [24]. ↵ Y. Gu et al. , “ Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing ,” ACM Trans. Comput. Healthc ., vol. 3 , no. 1 , pp. 1 – 24 , 2022 , doi: 10.1145/3458754 . OpenUrl CrossRef [25]. ↵ N. Reimers and I. Gurevych , “ Sentence-BERT: Sentence embeddings using siamese BERT-networks ,” EMNLP-IJCNLP 2019 - 2019 Conf. Empir. Methods Nat. Lang. Process. 9th Int. Jt. Conf. Nat. Lang. Process. Proc. Conf ., pp. 3982 – 3992 , 2019 , doi: 10.18653/v1/d19-1410 . OpenUrl CrossRef [26]. ↵ G. Huang , Z. Liu , L. van der Maaten , and K. Q. Weinberger , “ Densely Connected Convolutional Networks ,” in CVPR , 2017 , pp. 4700 – 4708 . [27]. ↵ X. Zhai et al. , “ LiT: Zero-Shot Transfer with Locked-image text Tuning ,” Proc. IEEE Comput. Soc. Conf. Comput. Vis. Pattern Recognit ., vol. 2022-June , pp. 18102 – 18112 , 2022 , doi: 10.1109/CVPR52688.2022.01759 . OpenUrl CrossRef [28]. ↵ R. Nogueira and K. Cho , “ Passage Re-ranking with BERT ,” arXiv , pp. 1 – 5 , 2019 , [Online]. Available: http://arxiv.org/abs/1901.04085 . [29]. ↵ R. Pradeep , Y. Liu , X. Zhang , Y. Li , A. Yates , and J. Lin , Squeezing Water from a Stone: A Bag of Tricks for Further Improving Cross-Encoder Effectiveness for Reranking , vol. 13185 LNCS. Springer International Publishing , 2022 . [30]. ↵ Gemma Team et al. , “ Gemma 2: Improving Open Language Models at a Practical Size ,” arXiv , 2024 , [Online]. Available: http://arxiv.org/abs/2408.00118 . [31]. ↵ A. Dubey et al. , “ The Llama 3 Herd of Models ,” arXiv , 2024 , [Online]. Available: http://arxiv.org/abs/2407.21783 . [32]. ↵ A. Q. Jiang et al. , “ Mistral 7B ,” arXiv , 2023 , [Online]. Available: http://arxiv.org/abs/2310.06825 . [33]. ↵ M. Douze et al. , “ The Faiss library ,” arXiv , 2024 , [Online]. Available: http://arxiv.org/abs/2401.08281 . View the discussion thread. Back to top Previous Next Posted February 20, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Leveraging a Vision-Language Model with Natural Text Supervision for MRI Retrieval, Captioning, Classification, and Visual Question Answering Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Leveraging a Vision-Language Model with Natural Text Supervision for MRI Retrieval, Captioning, Classification, and Visual Question Answering Nikhil J. Dhinagar , Sophia I. Thomopoulos , Paul M. Thompson bioRxiv 2025.02.15.638446; doi: https://doi.org/10.1101/2025.02.15.638446 Share This Article: Copy Citation Tools Leveraging a Vision-Language Model with Natural Text Supervision for MRI Retrieval, Captioning, Classification, and Visual Question Answering Nikhil J. Dhinagar , Sophia I. Thomopoulos , Paul M. Thompson bioRxiv 2025.02.15.638446; doi: https://doi.org/10.1101/2025.02.15.638446 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Neuroscience Subject Areas All Articles Animal Behavior and Cognition (7624) Biochemistry (17651) Bioengineering (13871) Bioinformatics (41882) Biophysics (21424) Cancer Biology (18566) Cell Biology (25461) Clinical Trials (138) Developmental Biology (13365) Ecology (19867) Epidemiology (2067) Evolutionary Biology (24290) Genetics (15590) Genomics (22476) Immunology (17714) Microbiology (40331) Molecular Biology (17148) Neuroscience (88483) Paleontology (666) Pathology (2828) Pharmacology and Toxicology (4817) Physiology (7635) Plant Biology (15114) Scientific Communication and Education (2044) Synthetic Biology (4286) Systems Biology (9815) Zoology (2268)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00