Full text
57,178 characters
· extracted from
preprint-html
· click to expand
DrugLM: A Unified Framework to Enhance Drug-Target Interaction Predictions by Incorporating Textual Embeddings via Language Models | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results DrugLM: A Unified Framework to Enhance Drug-Target Interaction Predictions by Incorporating Textual Embeddings via Language Models View ORCID Profile Tianyi Li , View ORCID Profile Zhengyu Fang , View ORCID Profile Xiaoge Zhang , View ORCID Profile Kaiyu Tang , View ORCID Profile Huiyuan Chen , View ORCID Profile Zhimeng Jiang , View ORCID Profile Tianxiang Zhao , View ORCID Profile Rong Xu , View ORCID Profile Feixiong Cheng , View ORCID Profile Xiao Li , View ORCID Profile Jing Li doi: https://doi.org/10.1101/2025.07.09.657250 Tianyi Li 1 Department of Computer and Data Sciences, Case Western Reserve University , 10900 Euclid Ave, 44106, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tianyi Li Zhengyu Fang 1 Department of Computer and Data Sciences, Case Western Reserve University , 10900 Euclid Ave, 44106, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Zhengyu Fang Xiaoge Zhang 1 Department of Computer and Data Sciences, Case Western Reserve University , 10900 Euclid Ave, 44106, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Xiaoge Zhang Kaiyu Tang 1 Department of Computer and Data Sciences, Case Western Reserve University , 10900 Euclid Ave, 44106, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kaiyu Tang Huiyuan Chen 1 Department of Computer and Data Sciences, Case Western Reserve University , 10900 Euclid Ave, 44106, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Huiyuan Chen Zhimeng Jiang 2 Department of Computer Science & Engineering, Texas A&M University , College Station, 77843, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Zhimeng Jiang Tianxiang Zhao 3 Department of Computer Science and Engineering, Pennsylvania State University , University Park, 16802, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tianxiang Zhao Rong Xu 4 Center for Artificial Intelligence in Drug Discovery, Case Western Reserve University , 10900 Euclid Ave, 44106, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Rong Xu Feixiong Cheng 5 Genomic Medicine Institute, Cleveland Clinic , 9500 Euclid Ave, 44195, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Feixiong Cheng Xiao Li 6 Department of Biochemistry and Center for RNA Science and Therapeutics, Case Western Reserve University , Cleveland, 10900 Euclid Ave, 44106, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Xiao Li Jing Li 1 Department of Computer and Data Sciences, Case Western Reserve University , 10900 Euclid Ave, 44106, OH, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jing Li For correspondence: jingli{at}case.edu Abstract Full Text Info/History Metrics Preview PDF Abstract Motivation Accurate prediction of drug–target interactions (DTIs) is central to computational drug discovery, offering the potential to reduce experimental costs and accelerate development timelines. While existing deep learning approaches such as Graph Neural Networks and Transformers have shown promise, they often overlook the rich semantic information embedded in textual descriptions of drugs and targets. These descriptions encode critical biomedical knowledge, including mechanisms of action, biological pathways involved, and therapeutic effects of drugs, which can enhance DTI prediction performance. Results We introduce DrugLM , a unified framework that integrates embeddings derived from large language models (LLMs) into DTI-specific model architectures. DrugLM leverages textual descriptions of drugs and targets to generate semantic embeddings using a range of pretrained LLMs. These embeddings can be seamlessly incorporated into existing DTI models. We systematically evaluate multiple LLMs on benchmark DTI datasets and demonstrate strong performance even without fine-tuning. Moreover, supervised parameter-efficient fine-tuning of the LLMs further improves embedding quality, leading to enhanced prediction accuracy. Notably, a simple multilayer perceptron (MLP) using only LLM-derived embeddings surpasses several established DTI methods, underscoring the power of semantic features. Our findings highlight the practical value of integrating LLMs into DTI pipelines and offer a straightforward recipe for improved drug discovery: LLM embeddings of drugs and targets are both effective and easy to use. Availability Our code and dataset are available at https://github.com/ShPhoebus/DrugLM 1 Introduction Traditional drug discovery is notoriously expensive, time-consuming, and prone to high failure rates ( Mullard, 2014 ; Menden et al ., 2019 ). Estimates suggest that the development of a single drug can cost between$314 million and$2.8 billion, with clinical phases spanning 8.2 to 10 years ( Brown et al ., 2021 ). To mitigate these challenges and reduce dependence on resource-intensive laboratory procedures, substantial efforts have been directed toward computational approaches, particularly those leveraging deep learning ( Gawehn et al ., 2015 ; Chen and Li, 2019b , Chen and Li, 2018 ). Among various computational drug discovery tasks, drug–target interaction (DTI) prediction plays a foundational role, offering mechanistic insights into pharmacological interventions ( Wu et al ., 2018 ; Chen et al ., 2020a ; Xiong et al ., 2021 ). Recent deep learning frameworks have achieved impressive results in DTI prediction ( Jones et al ., 2021 ; Chen and Li, 2020 ), especially those based on Graph Neural Networks (GNNs) ( Yang et al ., 2022 ; Xu et al ., 2023 , 2025 ; Fang et al ., 2025 ) and Transformer-based models ( Huang et al ., 2020 ). For instance, GraphDTA (Nguyen et al ., 2020) represents drug chemical structures as graphs and employs GNNs to estimate drug–target affinity. Similarly, MolTrans ( Huang et al ., 2020 ) adopts a Transformer-based architecture to capture interactions between molecular substructures, using linear representations of SMILES strings and amino acid sequences. While these models are effective, they primarily rely on data from a single modality: either one-dimensional string-based representations or two-dimensional graphs. They often overlook information from other modalities, such as three-dimensional structural information, or textual descriptions. More recently, there is a growing interest in integrating multimodal biological data, including three-dimensional structural information ( Liu et al ., 2025 ) and textual descriptions ( Vazquez et al ., 2011 ; Chen and Li, 2019a ; Tarca et al ., 2021 ), to further enhance DTI prediction performance. Recently, language models (LMs) have emerged as powerful tools for applications in biology and chemistry ( Guo et al ., 2023 ; Lin et al ., 2023 ; Hie et al ., 2023 ; Zhao et al ., 2025 ). For instance, BioT5 integrates molecular string representations with protein names, sequences, and structural data to train LMs on diverse prediction tasks ( Pei et al ., 2023 ). Similarly, DTI-LM employs LMs to encode protein amino acid sequences and drug SMILES strings for DTI prediction ( Ahmed et al ., 2024 ). However, existing approaches often overlook the rich semantic information embedded in unstructured textual descriptions of drugs and targets. These descriptions encapsulate critical biomedical knowledge, such as mechanisms of action, biological pathways involved, and therapeutic effects, which could potentially improve the accuracy of DTI prediction ( Singhal et al ., 2023 ). In this work, we aim to incorporate prior knowledge of drugs and protein targets to facilitate DTI prediction. To this end, we propose DrugLM , a unified framework that integrates embeddings derived from large language models (LLMs) into DTI-specific architectures. DrugLM follows a two-stage design. In the first stage, it generates semantic embeddings of drugs and targets using their free-text descriptions and multiple pretrained LLMs. For this work, we select three LMs from the top performers on the MTEB leader board 1 . In the second stage, these embeddings are seamlessly incorporated into downstream DTI prediction backbones. We include five widely used DTI prediction models (DeepConv-DTI ( Lee et al ., 2019 ), GraphDTA ( Nguyen et al ., 2020 ), NGCF ( Wang et al ., 2019 ), LightGCN ( He et al ., 2020 ), BACPI ( Li et al ., 2022 )) and evaluate their performance with the three pretrained LMs on a curated drug–target dataset. Results show that they all perform better than the original models in terms of AUC and accuracy. Additionally, we apply parameter-efficient supervised fine-tuning on these LLM models, resulting in refined embeddings and notable improvements in DTI prediction accuracy. These results demonstrate the value of text-derived semantic features and the potential of LLMs to enhance DTI prediction pipelines. In summary, our main contributions are as follows: Incorporating textual semantics improves DTI prediction. We demonstrate that integrating rich textual descriptions of drugs and targets significantly boosts the performance of various DTI models. Specifically, incorporating pretrained LM embeddings into the BACPI framework results in a 38.48% improvement over its baseline performance. Similarly, applying this approach to the GraphDTA framework yields a 12.21% increase in prediction accuracy. Parameter-efficient fine-tuning enhances embedding quality We show that supervised, parameter-efficient fine-tuning of language models leads to superior results compared to using pretrained models directly. This approach improves embedding quality via accelerated training. For example, within the GraphDTA framework, fine-tuning delivers an additional 8.96% performance gain over the non-fine-tuned counterpart. DrugLM is a flexible and effective solution for DTI Our proposed DrugLM framework can be seamlessly integrated into any existing DTI backbone. Notably, a simple MLP-based architecture using DrugLM embeddings outperforms several well-established DTI models. We offer a practical and effective recipe for DTI prediction: LLM-derived embeddings of drugs and targets are surprisingly effective for drug discovery . 2 Related Works In this section, we review key computational approaches for drug–target interaction prediction and highlight recent advances in language models that have been applied to drug discovery tasks. 2.1 Computational Drug-Target Prediction Numerous computational techniques have been developed to address the challenge of DTI prediction, aiming to reduce the high costs and time demands of experimental methods ( Lee et al ., 2019 ; Bai et al ., 2023 ; Lim et al ., 2019 ; Chen et al ., 2020b , a ; Chen and Li, 2019b ; Nguyen et al ., 2020 ; Yang et al ., 2024 ; Wang et al ., 2024 , Wang et al ., 2023b ; Yan et al ., 2023 ). Among these, deep learning approaches have gained significant attention for their ability to automatically extract informative features from raw inputs and capture complex non-linear relationships. For example, DeepConv-DTI ( Lee et al ., 2019 ) employs convolutional neural networks (CNNs) to learn local patterns from protein sequences and drug SMILES representations. GraphDTA ( Nguyen et al ., 2020 ) represents drug molecules as graphs and leverages graph neural networks (GNNs) to better capture topological and structural properties. These deep models have consistently outperformed traditional techniques such as matrix factorization ( Chen et al ., 2020a ), owing to their capacity for hierarchical feature learning. Graph-based collaborative filtering methods model drugs and targets as nodes in a bipartite interaction graph and learn their embeddings by propagating information across edges. Neural Graph Collaborative Filtering (NGCF) ( Wang et al ., 2019 ) injects both first- and higher-order collaborative signals into the embedding process via layer-wise message passing with nonlinear feature interaction, explicitly capturing complex connectivity patterns in the drug–target graph. LightGCN ( He et al ., 2020 ) simplifies this approach by removing feature transformations and activation functions, relying solely on normalized neighborhood aggregation and layer combination. More recently, attention mechanisms and Transformer-based architectures have been introduced to further improve model expressiveness and interpretability. Bi-directional Attention-based Compound–Protein Interaction (BACPI) ( Li et al ., 2022 ) integrates a graph attention network to encode compound structures and a convolutional neural network to encode protein sequences, fusing them via a bi-directional attention module that lets atom- and residue-level features attend to each other. DrugBAN ( Bai et al ., 2023 ) incorporates a bilinear attention network to focus on meaningful drug substructures and protein residues, enhancing both prediction accuracy and biological interpretability. MolTrans ( Huang et al ., 2020 ) adopts a self-attention framework to model interactions between subcomponents of drugs and proteins, treating DTI prediction as a sequence-to-sequence translation task. TransformerCPI ( Chen et al ., 2020b ) uses a Transformer encoder to capture contextual relationships within protein sequences and drug representations, achieving strong generalization to unseen interactions. Despite these advances, most existing models overlook the rich semantic information embedded in textual descriptions of drugs and targets. Such descriptions often encode valuable biomedical knowledge, including mechanisms of action, biological pathways involved, and therapeutic effects ( Wang et al ., 2023a ; Zheng et al ., 2024 ), which, if properly leveraged, could significantly enhance DTI prediction performance. 2.2 Language Models for Drug Discovery LLMs have recently demonstrated remarkable capabilities in understanding scientific language and supporting a wide range of downstream tasks critical to drug discovery and development ( Zheng et al ., 2024 ; Nguyen and Grover, 2025 ; Wang et al ., 2025 ). A prominent example is Geneformer ( Theodoris et al ., 2023 ), an LLM trained on 30 million single-cell transcriptomes. Geneformer has shown potential in disease modeling and was able to identify candidate therapeutic targets for cardiomyopathy through in silico gene deletion experiments. In the context of DTI prediction, DTI-LM ( Ahmed et al ., 2024 ) employs ESM-2 ( Lin et al ., 2023 ) to encode protein amino acid sequences and ChemBERTa ( Chithrananda et al ., 2020 ) to represent drug SMILES strings, demonstrating how specialized protein and molecular LMs can be integrated for end-to-end drug–target interaction modeling. In contrast to these approaches, our proposed DrugLM directly leverages large language models to derive embeddings from natural language descriptions of both drugs and targets. This enables a more flexible and semantically enriched representation, potentially enhancing the accuracy and interpretability of DTI predictions. 3 Materials and methods 3.1 Problem Setup We formulate DTI prediction as a binary classification task, where the goal is to determine whether a given drug–target protein pair is likely to interact. Let the set of drugs be denotedSS by 𝒟 = {d 1 , d 2 , …, d m } and the set of target proteins by 𝒫 = {p 1 , p 2 , …, p n } , where m and n are the total numbers of drugs and targets, respectively. The drugs and targets can have additional information associated with them, for examples, drug chemical structures, target sequences and structures. With a slight abuse of notation, we assume that such information is encoded in 𝒟 and 𝒫. We further assume that each drug is associated with a natural language textual description, denoted as . Similarly, each protein is also associated with a textual description, denoted as . These descriptions are free text in natural language and can encapsulate rich biomedical knowledge, including mechanisms of action, biological pathways involved, and therapeutic effects, which we hypothesize can enhance the predictive power of DTI models. The DTI prediction task is thus defined as learning a function or model ℱ that maps a drug–target pair, potentially enriched by their corresponding textual descriptions, to an interaction probability: where Ŷ ij represents the predicted probability of interaction between drug d i and target p j . 3.2 The Proposed Method: DrugLM We present a two-stage framework, DrugLM , which decouples the generation of LM embeddings from the actual DTI prediction task. This modular design provides two primary benefits: (1) it facilitates semantic alignment between drug and target representations via their textual descriptions, and (2) it introduces architectural flexibility, allowing seamless integration with diverse DTI prediction models. An overview of the framework is illustrated in Figure 1 . Download figure Open in new tab Fig. 1: Overview of our two-stage framework for DTI prediction. We first collect drug-target interaction data along with textual descriptions from DrugBank. In Stage 1, we train a language model (with or without fine-tuning) to generate embeddings for drugs and targets based on their textual descriptions. In Stage 2, these textual embeddings are integrated into downstream models such as DeepConv-DTA, BACPI, GraphDTA, NGCF, LightGCN, or a simple MLP. 3.2.1 Data Collection We utilize DrugBank 2 , a comprehensive biomedical database containing detailed molecular, pharmacological, and interaction information for both approved and experimental drugs. In addition to molecular profiles, DrugBank provides rich annotations on drug–target interactions, therapeutic indications, and related biological pathways—making it a valuable resource for tasks such as DTI prediction, drug property prediction and bioactivity modeling. For each drug, we extract its associated textual description and metadata, including therapeutic use, pharmacological mechanism, pharmacokinetic properties, and physicochemical characteristics, from DrugBank. To avoid information leaks, direct drug-target relationships were not included in the text description. For each target protein, we retrieve the corresponding natural language information from UniProt 3 . To ensure compatibility with general-purpose LMs, we apply the following preprocessing steps: Text Concatenation: For each drug and each target, we concatenate all available textual description into a single unified text representation suitable for language model encoding. Exclusion of Non-Natural Language Elements: Structured data such as SMILES strings (for drugs), InChI identifiers, amino acid sequences, and gene sequences (for proteins) are removed. These formats are not directly interpretable by general-purpose LMs and may degrade model performance if not appropriately handled. As a result, each drug and each target is associated with a unified natural language description, collectively denoted as 𝒯 d for drugs and 𝒯 p for targets as defined earlier. These textual representations form the input for the subsequent LM-based embedding generation. 3.2.2 Language Model Embeddings Textual descriptions of drugs and targets encode rich biological information relevant to drug–target interactions. To harness this information, we employ two strategies for obtaining language model-based embeddings: (1) utilizing pre-trained models without fine-tuning, and (2) performing task-specific fine-tuning to better align representations with the DTI prediction objective. Pre-trained LM Embeddings Without Fine-tuning We begin by leveraging language models that have been pretrained on large-scale biomedical corpora. These models are assumed to encode deep semantic knowledge, including biomedical terminology, functional relationships and semantics. By directly applying these models, we extract embeddings from the natural language descriptions of drugs and proteins without requiring any additional parameter updates. Formally, given the textual descriptions and for drug i and protein j , and a pre-trained language model LM( · ), the corresponding semantic embeddings are computed as: where and denote the embedding vectors for drug i and target j , respectively. LM Embeddings with Fine-tuning Fine-tuning a language model refers to the process of adapting a pre-trained model to a specific downstream task by continuing its training on a smaller, domain-specific dataset ( Hu et al ., 2022 ). In this work, we explore whether fine-tuning can enhance the ability of LMs to capture nuanced relationships between drugs and targets. We fine-tune the LM using a contrastive learning objective tailored to the DTI prediction task. Specifically, we minimize the contrastive cross-entropy loss: where sim( ·, · ) denotes the cosine similarity between embeddings, is the positive target embedding that interacts with drug i , and are negative (non-interacting) target embeddings. Here, τ is the temperature parameter and θ represents the learnable parameters of the LM. We set N = 4: since the dataset contains only positive interaction pairs, we randomly sample four negative target embeddings for each positive pair from all non-interacting targets. We sum over i to aggregate the contrastive loss across all positive drug–target pairs. Although we treat all unknown pairs as negatives, one could further refine this by hard-negative mining or by excluding known off-target interactions. If a drug has multiple true targets, each ( d, p ) pair is treated separately in the above summation. After fine-tuning, the resulting embeddings are tailored to the DTI prediction task. Fine-tuning Strategy To ensure both computational efficiency and training stability, we adopt a supervised, parameter-efficient fine-tuning strategy. Specifically, we freeze the lower layers of the transformer, which primarily capture general linguistic patterns, and fine-tune only the upper layers that are more task-specific 4 . Formally, let the parameters of the language model be denoted by θ . We partition them as follows: where θ frozen corresponds to the parameters of the lower (frozen) layers, and θ tuned denotes the parameters of the upper layers, which are fine-tuned for the DTI prediction task. 3.2.3 Selection of Pretrained LMs In this study, we select candidate language models from the MTEB leaderboard, prioritizing models that have been pretrained for semantic retrieval tasks. Considering factors such as model size, architecture, and performance on both retrieval and classification benchmarks, we primarily focus on three high-performing models: BGE-large-en-v1.5, GTE-large-en-v1.5 , and E5-large-v2 . BGE-large-v1.5 leverages RetromAE pre-training and large-scale contrastive fine-tuning on text pairs to alleviate similarity distribution imbalance and provide robust retrieval embeddings ( Xiao et al ., 2024 ). GTE-large-en-v1.5 builds on a Transformer++ encoder (BERT + RoPE + GLU) to support long-context embedding of up to 8 192 tokens, producing robust representations for extended input sequences ( Zhang et al ., 2024 ; Li et al ., 2023 ). E5-large-en-v2 uses a 24-layer Transformer encoder pre-trained with weakly supervised contrastive learning on large-scale text pairs, yielding robust single-vector embeddings for both retrieval and classification tasks ( Wang et al ., 2022 ). Throughout the paper, we simply use BGE, GTE and E5 to denote these models. Embedding Extraction Methods Different LM architectures require distinct strategies for extracting embeddings. For models such as GTE and BGE, we extract the representation from the special classification token (), which is a dedicated placeholder inserted at the very start of each input sequence, whose final-layer hidden state is trained to capture a holistic summary of the entire input. In contrast, for the E5 model, we apply average pooling over all token embeddings to obtain the final representation. Additionally, we apply the model-specific prompting templates recommended by each model’s developers to improve embedding quality. Specifically, for E5 we prepend “query: “to drug texts and “passage:” to target-protein descriptions; for BGE we add “Represent this sentence:” before every input (drugs and proteins alike); and for GTE we use the raw text directly, without any prefix. 3.2.4 DTI Prediction Model Selection Our proposed DrugLM framework is model-agnostic and can be seamlessly integrated into any downstream DTI prediction architecture. By leveraging language models, the resulting embeddings are expected to effectively capture semantic features aligned with biological mechanisms and functional properties relevant to drug–target interactions. These LM-derived embeddings are incorporated as complementary features to enhance the performance of various DTI prediction models. In this work, we primarily focus on five widely adopted model architectures for the DTI task. In addition, we also adopt a simple MLP model as another baseline to evaluate the performance improvement when incorporating the LM-derived embeddings. We briefly summarize the specific strategies we adopt for each of the DTI prediction models here. DeepConv-DTI ( Lee et al ., 2019 ): DeepConv-DTI takes the binary fingerprint representation for drug i and the protein sequence for target j as its inputs. It employs fully connected layers to learn a latent drug representation: and adopts a convolutional neural network to detect localized residue patterns of target: . To incorporate textual information, we simply concatenate their original latent representations with LM-derived embeddings (from Eq. (1) or Eq. (2) ) as follows: where and are the final representations of drug i and target j , respectively. These representations are concatenated and passed through fully connected layers to predict the interaction likelihood. GraphDTA ( Nguyen et al ., 2020 ): Similar to DeepConv-DTI, GraphDTA employs CNNs to learn protein sequence representations. However, instead of using fully connected layers for drug binary fingerprint, it models drugs as graphs and applies graph neural networks (GNNs) to learn their representations: . We integrate LM embeddings by concatenating them as follows: NGCF ( Wang et al ., 2019 ) and LightGCN ( He et al ., 2020 ): Both approaches were initially developed for recommender systems to impute missing links between users and items. They can equally be applied to the DTI prediction problem by representing drug–target interactions as a bipartite graph and utilizing GNNs to learn embeddings for both drugs and targets by propagating information through the graph structure. Because their original work didnot consider the DTI application, we modify their implementations slightly by considering two alternatives for the node embedding initialization: (1) random initialization as a baseline, and initialization with the LM-based textual embeddings. This comparison allows us to assess the impact of semantic information on DTI prediction. BACPI ( Li et al ., 2022 ): BACPI employs a graph attention network for drugs and a convolutional neural network for targets. Following the same strategy as in DeepConv-DTI and GraphDTA, we augment the original representations with our text-based embeddings. These combined embeddings are then fed into a bidirectional attention network to model drug–target interactions. MLP : As a simple yet informative baseline, we propose a multilayer perceptron (MLP) model that directly utilizes our LM-based textual embeddings. Given the drug embedding and target embedding , we concatenate them and use the MLP to predict the interaction: where y ij denotes the predicted probability of interaction between drug i and target j . This setup allows us to evaluate whether semantic information alone is sufficient for effective DTI prediction. 4 Experiments In this section, we evaluate the performance of DrugLM on real-world datasets. We curated a dataset consisting of 13,924 validated drug–target interactions, involving 6,673 unique drugs and 3,890 distinct targets. The data collection and preprocessing procedures are detailed in Section 3.2.1 . To facilitate model training and evaluation, we employed a stratified edge-based splitting strategy: for drugs with fewer than 10 associated targets, all their edges were placed in the training set; for drugs with 10 or more targets, 80% of their edges were used for training and 20% for testing. Finally, 20% of the combined training edges were randomly held out as a validation set. The training set was used to train both the proposed models and the LM-based embeddings. The validation set was used for hyperparameter tuning and performance monitoring during training, while the test set was held out to ensure an unbiased evaluation of the final model performance. To ensure fair comparison, SMILES-based baselines (DeepConv-DTI, GraphDTA, BACPI) were evaluated only on drugs with valid SMILES (excluding biologics), whereas LightGCN, NGCF and MLP were applied to both small molecules and biologics. Following established protocols from prior works ( Huang et al., 2020 ; Lee et al., 2019 ), we adopt some standard classificationbased metrics for the binary prediction task. Specifically, we report three commonly used metrics: Accuracy (ACC), Area Under the Receiver Operating Characteristic Curve (AUC), and Area Under the Precision–Recall Curve (AUPR). To ensure robustness and reproducibility, all experiments were repeated five times with different random seeds. The final results are reported as the average of these five independent runs, providing a reliable estimate of the model’s stability and generalizability. Additionally, we performed paired two-sided t-tests on the five independent runs to assess statistical significance of embedding improvements. 4.1 Effectiveness of DrugLM We present a comprehensive comparative analysis of various embedding strategies and model architectures for DTI prediction. Table 1 summarizes the overall results in terms of AUC, ACC, and AUPR. Our findings underscore the substantial impact by including language embeddings. Key observations are outlined below: View this table: View inline View popup Download powerpoint Table 1. Performance of different language model embeddings on DTI prediction (mean±standard error over five independent runs). Bold values highlight the best results for each DTI model. FT: fine-tuned. Significance (paired t-test vs baseline): * p < 0.05, ** p < 0.01 (baseline = Original Embedding for DeepConv-DTI, GraphDTA, BACPI; Random Embedding for NGCF, LightGCN, MLP). Adding LLM embeddings significantly improve the performance of all models Across all tested DTI prediction models, incorporating LLM embeddings (E5, GTE, BGE) consistently results in significant performance gains compared to the models using their original representations alone, or randomly initialized embeddings. The scales of improvements vary depending on the DTI models, the LMs used in generating the embedding, and pre-trained or fine-tuned embeddings. Across all the experiments and all the evaluation metrics, BGE with fine-tuning achieved the best results. LLM embeddings with fine-tuning consistently boosts performance: Fine-tuned LLM embeddings outperform their non-fine-tuned counterparts across various DTI models, reinforcing the effectiveness of domain adaptation via fine-tuning. For instance, using E5-FT with GraphDTA increases the AUC from 0.7349 to 0.8033 (a relative 9.31% gain) and the AUPR from 0.7105 to 0.7886 (a 10.99% gain) over non-fine-tuned E5. GNNs benefit from bipartite drug–target graph topology: Models that explicitly exploit the bipartite structure of drug–target interactions, such as NGCF and LightGCN, tend to outperform sequence-based and structure-based methods like DeepConv-DTI and GraphDTA, particularly when paired with high-quality embeddings. For example, LightGCN combined with BGE-FT achieves an AUC of 0.9098, substantially higher than DeepConv-DTI (0.7518) and GraphDTA (0.8139), highlighting the benefit of graph-based structural modeling in DTI tasks. Potential knowledge conflict with LLM embeddings: Interestingly, although fine-tuned embeddings generally enhance performance, some of them show a slight decline when used with certain DTI models such as DeepConv-DTI. For instance, E5-FT yields an AUPR of 0.7261, a 0.53% drop from the original E5’s AUPR of 0.7300. This suggests a potential conflict or redundancy between the contextual knowledge captured by LLMs and the handcrafted features (e.g., SMILES or amino acid sequences) utilized by the DeepConv-DTI architecture. Nonetheless, the general benefits of LLM embeddings remain substantial. DrugLM significantly enhances even simple MLPs: A particularly notable result is the strong performance of a simple multi-layer perceptron (MLP) when augmented with our LLM embeddings. Despite its simplicity, the MLP model achieves competitive and often superior results compared to more complex architectures. For example, MLP with GTE-FT achieves an AUC of 0.8811, outperforming DeepConv-DTI with BGE-FT (0.7463), GraphDTA with E5-FT (0.8033), and even BACPI with with GTE-FT (0.8727). This result demonstrates that high-quality language model embeddings can dramatically boost the performance of even basic models, offering a compelling and computationally efficient alternative for the DTI prediction task. In summary, our experiments show that DrugLM, through the use of advanced language model embeddings for both drugs and targets, provides a powerful framework for improving the DTI prediction performance. The substantial gains observed across diverse prediction architectures, including the very simple MLP model, underscore the value of leveraging rich, context-aware textual representations. 4.2 Visualization of Drug Embeddings To further investigate the effectiveness of LM embeddings, we compare the embeddings generated by LightGCN under three initialization strategies: Random initialization, BGE-NonFT (BGE pre-trained without fine-tuning), and BGE-FT (BGE fine-tuned). Figure 2 shows the t-SNE projections of these embeddings of the three cases. These plots show the node embeddings after LightGCN training, where initial vectors are refined via neighborhood aggregation on the drug–target graph. To generate them, we first obtained initial embeddings under each initialization, then trained LightGCN and applied t-SNE to the resulting node vectors (panels a–c). All maps include 6,238 small-molecule drugs (blue dots) and 435 biotech drugs (orange triangles). The labels whether a drug was a small molecule drug or a biotech drug were not included in the original text description. Download figure Open in new tab Fig. 2: t-SNE visualization of drug embeddings generated by LightGCN under three initialization strategies: (a) Random , (b) BGE-NonFT (pre-trained BGE without fine-tuning), and (c) BGE-FT (fine-tuned BGE). Orange triangles indicate biotech drugs and blue circles indicate small-molecule drugs. A key observation is that the final embeddings under the BGE initializations (BGE-NonFT and BGE-FT) exhibit much clearer separation between small-molecule and biotech drugs than those with random initialization. In panel (a), Random embeddings display substantial overlap between the two classes, whereas panels (b) and (c) show progressively distinct clusters. This pronounced clustering demonstrates that language models capture domain-specific information in the embedding space and that graph-based refinement further accentuates these semantic distinctions. Such well-separated groupings are critical for precise DTI prediction and underscore the value of LM-based representations combined with graph learning for accelerating drug discovery. 4.3 Training Effectiveness To evaluate the training dynamics and effectiveness of DrugLM, we tracked the validation accuracy of all downstream DTI prediction backbones for 30 epochs under three initialization schemes: Random/Original initialization, BGE-NonFT (BGE pre-trained without fine-tuning), and BGE-FT (BGE fine-tuned). Figure 3 presents the validation accuracy curves for the LightGCN (a) and MLP (b) backbones. Analogous curves for DeepConv-DTI, GraphDTA, NGCF, and BACPI are provided in Supplementary Figure S1 . These curves illustrate the downstream DTI training process rather than the language model pre-training phase. Note that for LightGCN and NGCF, validation metrics were recorded only every five epochs in their original implementations, whereas all other models report results at every epoch. Download figure Open in new tab Fig. 3: Validation accuracy across 30 epochs for LightGCN (a) and MLP (b). Three embeddings were used: Random initialization, BGE-NonFT (BGE pre-trained without fine-tuning), and BGE-FT (fine-tuned BGE). Initializing DrugLM with pre-trained embeddings consistently yielded higher validation accuracy than random initialization for all models, highlighting the value of leveraging pretrained linguistic knowledge to create informative initial representations for drugs and targets. Fine-tuning the language model embeddings further amplified these gains by injecting additional domain-specific relevance, resulting in improved overall performance and more stable learning trajectories. We also observed that the scale of improvements varied greatly for different model structures, which could be explained based on how the models actually worked. For example, when using random initialization for MLP, no meaningful information was provided as inputs. Therefore, the accuracy for the prediction was always around 0.5 for the binary classification. For LightGCN, the random initialization was only used for the node annotations of the drug-target bipartite graph. Although the annotations did not provide any information, the topological structure of the drug-target bipartite graph can still allow the model to predict meaningful results. Notably, the magnitude of improvement varied substantially across different model structures, reflecting differences in how each model utilizes their own input representations. For instance, in the case of MLP, random initialization provided no meaningful input features, causing prediction accuracy to hover around 0.5 in this binary classification task. In contrast, for LightGCN, random initialization affected only the node features within the drug–target bipartite graph. Despite the lack of informative node annotations, LightGCN was still able to make meaningful predictions by leveraging the structural topology of the graph, demonstrating its relative robustness to feature initialization. Conclusion We introduced DrugLM , a unified framework that advances drug-target interaction prediction by integrating rich, text-driven semantic information from language models into existing prediction model architectures. The core premise of DrugLM is that the extensive biomedical knowledge encoded in textual descriptions of drugs and targets—including mechanisms of action and therapeutic effects—can substantially enhance DTI prediction accuracy. Comprehensive experiments demonstrated the effectiveness of DrugLM across various DTI prediction models. Leveraging LM-derived embeddings for drug and target representations consistently yielded significant performance gains. In particular, pre-trained LM embeddings outperformed random/original initialization, underscoring the value of linguistic prior knowledge. Moreover, parameter-efficient fine-tuning further improved both training efficiency and predictive performance by enhancing the domain-specific relevance of the embeddings. Remarkably, even a simple MLP model, when combined with LM embeddings, surpassed several specialized DTI architectures—highlighting the powerful semantic representation capabilities encoded in LMs. Beyond the LM models discussed, we also evaluated embeddings generated by a larger autoregressive language model, Llama3-7B . While modest improvements were observed, they were limited—likely due to the causal attention mechanism inherent to autoregressive models, which constrains their ability to produce fully contextualized representations ( Springer et al ., 2025 ). We leave the investigation of even larger models, such as Gemini and Llama3-80B , to future work. Overall, DrugLM offers a compelling and generalizable approach for incorporating advanced language models into DTI pipelines. By effectively leveraging unstructured biomedical text, DrugLM provides a simple yet powerful recipe for practical DTI prediction. The consistent benefits across diverse models and LMs suggest that DrugLM can substantially accelerate drug discovery by tapping into the wealth of semantic information embedded in biomedical literature. Supplementary Materials S1 Training Effectiveness Across Backbone Architectures and Different Embeddings Download figure Open in new tab Fig. S1: Validation accuracy across 30 epochs for different backbone architectures: (a) DeepConv-DTI, (b) GraphDTA, (c) NGCF, and (d) BACPI. For each model architecture, three embeddings were used: Random/Original initialization, BGE-NonFT (BGE pre-trained without fine-tuning), and BGE-FT (fine-tuned BGE). Footnotes ↵ 1 https://huggingface.co/spaces/mteb/leaderboard ↵ 2 https://go.drugbank.com/ ↵ 3 https://www.uniprot.org/ References ↵ Ahmed , K. T. et al. ( 2024 ). DTI-LM: language model powered drug–target interaction prediction . Bioinformatics , 40 ( 9 ). ↵ Bai , P. et al. ( 2023 ). Interpretable bilinear attention network with domain adaptation improves drug–target prediction . Nature Machine Intelligence , 5 ( 2 ), 126 – 136 . OpenUrl ↵ Brown , D. G. et al. ( 2021 ). Clinical development times for innovative drugs . Nature Reviews Drug Discovery , 21 ( 11 ), 793 – 794 . OpenUrl ↵ Chen , H. and Li , J. ( 2018 ). DrugCom: Synergistic discovery of drug combinations using tensor decomposition . In 2018 IEEE International Conference on Data Mining (ICDM) , page 899 – 904 . IEEE . ↵ Chen , H. and Li , J. ( 2019a ). Adversarial tensor factorization for context-aware recommendation . In Proceedings of the 13th ACM Conference on Recommender Systems, RecSys ‘19 , page 363 – 367 . ACM . ↵ Chen , H. and Li , J. ( 2019b ). Modeling relational drug-target-disease interactions via tensor factorization with multiple web sources . In The World Wide Web Conference, WWW ‘19 , page 218 – 227 . ACM . ↵ Chen , H. and Li , J. ( 2020 ). Learning data-driven drug-target-disease interaction via neural tensor network . In Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence, IJCAI-PRICAI-2020 , page 3452 – 3458 . International Joint Conferences on Artificial Intelligence Organization . ↵ Chen , H. et al. ( 2020a ). iDrug: Integration of drug repositioning and drug-target prediction via cross-network embedding . PLOS Computational Biology , 16 ( 7 ), e1008040 . OpenUrl ↵ Chen , L. et al. ( 2020b ). TransformerCPI: improving compound–protein interaction prediction by sequence-based deep learning with self-attention mechanism and label reversal experiments . Bioinformatics , 36 ( 16 ), 4406 – 4414 . OpenUrl CrossRef PubMed ↵ Chithrananda , S. et al. ( 2020 ). ChemBERTa: large-scale self-supervised pretraining for molecular property prediction . arXiv preprint arXiv: 2010.09885 . ↵ Fang , Z. et al. ( 2025 ). Recent developments in gnns for drug discovery . arXiv preprint arXiv: 2506.01302 . ↵ Gawehn , E. et al. ( 2015 ). Deep learning in drug discovery . Molecular Informatics , 35 ( 1 ), 3 – 14 . OpenUrl PubMed ↵ Guo , T. et al. ( 2023 ). What can large language models do in chemistry? a comprehensive benchmark on eight tasks . In Neural Information Processing Systems . ↵ He , X. et al. ( 2020 ). LightGCN: Simplifying and powering graph convolution network for recommendation . In Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR ‘20 , page 639 – 648 . ACM . ↵ Hie , B. L. et al. ( 2023 ). Efficient evolution of human antibodies from general protein language models . Nature Biotechnology , 42 ( 2 ), 275 – 283 . OpenUrl PubMed ↵ Hu , E. J. et al. ( 2022 ). LoRA: Low-rank adaptation of large language models . In ICLR . ↵ Huang , K. et al. ( 2020 ). MolTrans: Molecular interaction transformer for drug–target interaction prediction . Bioinformatics , 37 ( 6 ), 830 – 836 . OpenUrl CrossRef ↵ Jones , D. et al. ( 2021 ). Improved protein–ligand binding affinity prediction with structure-based deep fusion inference . Journal of Chemical Information and Modeling , 61 ( 4 ), 1583 – 1592 . OpenUrl CrossRef PubMed ↵ Lee , I. et al. ( 2019 ). DeepConv-DTI: Prediction of drug-target interactions via deep learning with convolution on protein sequences . PLOS Computational Biology , 15 ( 6 ), e1007129 . OpenUrl PubMed ↵ Li , M. et al. ( 2022 ). BACPI: a bi-directional attention neural network for compound–protein interaction and binding affinity prediction . Bioinformatics , 38 ( 7 ), 1995 – 2002 . OpenUrl CrossRef PubMed ↵ Li , Z. et al. ( 2023 ). Towards general text embeddings with multi-stage contrastive learning . arXiv preprint arXiv: 2308.03281 . ↵ Lim , J. et al. ( 2019 ). Predicting drug–target interaction using a novel graph neural network with 3d structure-embedded graph representation . Journal of Chemical Information and Modeling , 59 ( 9 ), 3981 – 3988 . OpenUrl CrossRef PubMed ↵ Lin , Z. et al. ( 2023 ). Evolutionary-scale prediction of atomic-level protein structure with a language model . Science , 379 ( 6637 ), 1123 – 1130 . OpenUrl CrossRef PubMed ↵ Liu , Z. et al. ( 2025 ). NExT-Mol: 3d diffusion meets 1d language modeling for 3d molecule generation . In The Thirteenth International Conference on Learning Representations . ↵ Menden , M. P. et al. ( 2019 ). Community assessment to advance computational prediction of cancer drug combinations in a pharmacogenomic screen . Nature Communications , 10 ( 1 ). ↵ Mullard , A. ( 2014 ). New drugs cost us 2.6 billion to develop . Nature reviews drug discovery , 13 ( 12 ), 877 – 877 . OpenUrl ↵ Nguyen , T. and Grover , A. ( 2025 ). LICO: Large language models for in-context molecular optimization . In The Thirteenth International Conference on Learning Representations . ↵ Nguyen , T. et al. ( 2020 ). GraphDTA: predicting drug–target binding affinity with graph neural networks . Bioinformatics , 37 ( 8 ), 1140 – 1147 . OpenUrl CrossRef ↵ Pei , Q. et al. ( 2023 ). BioT5: Enriching cross-modal integration in biology with chemical knowledge and natural language associations . In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing . Association for Computational Linguistics . ↵ Singhal , K. et al. ( 2023 ). Large language models encode clinical knowledge . Nature , 620 ( 7972 ), 172 – 180 . OpenUrl CrossRef PubMed ↵ Springer , J. M. et al. ( 2025 ). Repetition improves language model embeddings . In ICLR . ↵ Tarca , A. L. et al. ( 2021 ). Crowdsourcing assessment of maternal blood multi-omics for predicting gestational age and preterm birth . Cell Reports Medicine , 2 ( 6 ), 100323 . OpenUrl CrossRef PubMed ↵ Theodoris , C. V. et al. ( 2023 ). Transfer learning enables predictions in network biology . Nature , 618 ( 7965 ), 616 – 624 . OpenUrl CrossRef PubMed ↵ Vazquez , M. et al. ( 2011 ). Text mining for drugs and chemical compounds: Methods, tools and applications . Molecular Informatics , 30 ( 6–7 ), 506 – 519 . OpenUrl PubMed ↵ Wang , H. et al. ( 2025 ). Efficient evolutionary search over chemical space with large language models . In The Thirteenth International Conference on Learning Representations . ↵ Wang , L. et al. ( 2022 ). Text embeddings by weakly-supervised contrastive pre-training . arXiv preprint arXiv: 2212.03533 . ↵ Wang , R. et al. ( 2023a ). ChatGPT in drug discovery: A case study on anticocaine addiction drug development with chatbots . Journal of Chemical Information and Modeling , 63 ( 22 ), 7189 – 7209 . OpenUrl PubMed ↵ Wang , S. et al. ( 2023b ). Federated few-shot learning . In Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, KDD ‘23 , page 2374 – 2385 . ACM . ↵ Wang , S. et al. ( 2024 ). Enhancing distribution and label consistency for graph out-of-distribution generalization . In 2024 IEEE International Conference on Data Mining . IEEE . ↵ Wang , X. et al. ( 2019 ). Neural Graph Collaborative Filtering . In Proceedings of the 42nd International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR ‘19 , page 165 – 174 . ACM . ↵ Wu , Z. et al. ( 2018 ). MoleculeNet: a benchmark for molecular machine learning . Chemical Science , 9 ( 2 ), 513 – 530 . OpenUrl CrossRef PubMed ↵ Xiao , S. et al. ( 2024 ). C-Pack: Packed resources for general chinese embeddings . In Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2024 , page 641 – 649 . ACM . ↵ Xiong , Z. et al. ( 2021 ). Crowdsourced identification of multi-target kinase inhibitors for ret- and tau-based disease: The multi-targeting drug dream challenge . PLOS Computational Biology , 17 ( 9 ), e1009302 . OpenUrl PubMed ↵ Xu , Z. et al. ( 2023 ). Kernel ridge regression-based graph dataset distillation . In Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, KDD ‘23 , page 2850 – 2861 . ACM . ↵ Xu , Z. et al. ( 2025 ). Discrete-state continuous-time diffusion for graph generation . In Proceedings of the 38th International Conference on Neural Information Processing Systems, NIPS ‘24 , Red Hook, NY, USA . Curran Associates Inc . ↵ Yan , Y. et al. ( 2023 ). From trainable negative depth to edge heterophily in graphs . Advances in Neural Information Processing Systems , 36 , 70162 – 70178 . OpenUrl ↵ Yang , X. et al. ( 2024 ). SimCE: Simplifying cross-entropy loss for collaborative filtering . arXiv preprint arXiv: 2406.16170 . ↵ Yang , Z. et al. ( 2022 ). MGraphDTA: deep multiscale graph neural network for explainable drug–target binding affinity prediction . Chemical Science , 13 ( 3 ), 816 – 833 . OpenUrl CrossRef PubMed ↵ Zhang , X. et al. ( 2024 ). mGTE: Generalized long-context text representation and reranking models for multilingual text retrieval . arXiv preprint arXiv: 2407.19669 . ↵ Zhao , Y. et al. ( 2025 ). SimAug: Enhancing recommendation with pretrained language models for dense and balanced data augmentation . arXiv preprint arXiv: 2505.01695 . ↵ Zheng , Y. et al. ( 2024 ). Large language models in drug discovery and development: From disease mechanisms to clinical trials . arXiv preprint arXiv: 2409.04481 . View the discussion thread. Back to top Previous Next Posted July 11, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following DrugLM: A Unified Framework to Enhance Drug-Target Interaction Predictions by Incorporating Textual Embeddings via Language Models Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share DrugLM: A Unified Framework to Enhance Drug-Target Interaction Predictions by Incorporating Textual Embeddings via Language Models Tianyi Li , Zhengyu Fang , Xiaoge Zhang , Kaiyu Tang , Huiyuan Chen , Zhimeng Jiang , Tianxiang Zhao , Rong Xu , Feixiong Cheng , Xiao Li , Jing Li bioRxiv 2025.07.09.657250; doi: https://doi.org/10.1101/2025.07.09.657250 Share This Article: Copy Citation Tools DrugLM: A Unified Framework to Enhance Drug-Target Interaction Predictions by Incorporating Textual Embeddings via Language Models Tianyi Li , Zhengyu Fang , Xiaoge Zhang , Kaiyu Tang , Huiyuan Chen , Zhimeng Jiang , Tianxiang Zhao , Rong Xu , Feixiong Cheng , Xiao Li , Jing Li bioRxiv 2025.07.09.657250; doi: https://doi.org/10.1101/2025.07.09.657250 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7617) Biochemistry (17633) Bioengineering (13856) Bioinformatics (41841) Biophysics (21399) Cancer Biology (18529) Cell Biology (25422) Clinical Trials (138) Developmental Biology (13352) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24281) Genetics (15582) Genomics (22461) Immunology (17700) Microbiology (40293) Molecular Biology (17140) Neuroscience (88413) Paleontology (666) Pathology (2823) Pharmacology and Toxicology (4813) Physiology (7632) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4284) Systems Biology (9808) Zoology (2267)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.