Multimodal Molecular LLM with Improved Graph Utilization | Research Square window.SnipcartSettings = { analytics: { enabled: false } }; (function() { var accessVector = localStorage.getItem('access_vector') || ''; window.dataLayer = window.dataLayer || []; if (accessVector) { window.dataLayer.push({ user: { profile: { profileInfo: { snid: accessVector } } } }); } })(); (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src='https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-K279D39R'); Browse Preprints In Review Journals COVID-19 Preprints AJE Video Bytes Research Tools Research Promotion AJE Professional Editing AJE Rubriq About Preprint Platform In Review Editorial Policies Our Team Advisory Board Help Center Sign In Submit a Preprint Cite Share Download PDF Article Multimodal Molecular LLM with Improved Graph Utilization Sungwoong Kim, Chanhui Lee, Hanbum Ko, Yuheon Song, Yongjun Jeong, and 4 more This is a preprint; it has not been peer reviewed by a journal. https://doi.org/ 10.21203/rs.3.rs-8199982/v1 This work is licensed under a CC BY 4.0 License Status: Under Review Version 1 posted You are reading this latest preprint version Abstract Large language models (LLMs) are increasingly applied to molecular science, where instruction-tuning enables one model to address diverse tasks, complementing costly or narrow task-specific methods. Yet most molecular LLMs operate on molecular sequences like SMILES, leaving molecular structure reasoning implicit. Recent works augment molecular sequences by concatenating multimodal representations like 2D graphs. However, we observe that naive instruction-tuning through next-token prediction causes models to bypass graph information, including bond connectivity and substructure, which provide important cues for solving diverse molecular tasks. To resolve this, we introduce Molecular structure Preference Optimization (MolPO), a training objective where an LLM learns to generate answers by conditioning on correct graphs rather than perturbed ones. Furthermore, we introduce a hybrid graph encoder to enhance graph utilization, and leverage extensive instruction-tuning through staged training. The resulting model, Mol-LLM, demonstrates strong performance across broader tasks than prior generalist molecular LLMs, including robust out-of-distribution generalization where they degrade. Physical sciences/Chemistry Physical sciences/Chemistry/Chemical synthesis Physical sciences/Chemistry/Cheminformatics Physical sciences/Chemistry/Organic chemistry Physical sciences/Chemistry/Organic chemistry/Structure elucidation Molecular LLM Multimodal LLM Foundation model Preference optimization Generalization Full Text Additional Declarations There is NO Competing Interest. Cite Share Download PDF Status: Under Review Version 1 posted You are reading this latest preprint version Research Square lets you share your work early, gain feedback from the community, and start making changes to your manuscript prior to peer review in a journal. As a division of Research Square Company, we’re committed to making research communication faster, fairer, and more useful. We do this by developing innovative software and high quality services for the global research community. Our growing team is made up of researchers and industry professionals working together to solve the most critical problems facing scientific publishing. Also discoverable on Platform About Our Team In Review Editorial Policies Advisory Board Help Center Resources Author Services Accessibility API Access RSS feed Manage Cookie Preferences © Research Square 2026 | ISSN 2693-5015 (online) Privacy Policy Terms of Service Do Not Sell My Personal Information {"props":{"pageProps":{"initialData":{"identity":"rs-8199982","acceptedTermsAndConditions":true,"allowDirectSubmit":false,"archivedVersions":[],"articleType":"Article","associatedPublications":[],"authors":[{"id":630414605,"identity":"eafe7ab5-7fa8-4438-b282-080cccd1a9dd","order_by":0,"name":"Sungwoong Kim","email":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAZAAAAAyAQMAAABI0h/eAAAABlBMVEX///8AAABVwtN+AAAACXBIWXMAAA7EAAAOxAGVKw4bAAAA00lEQVRIiWNgGAWjYFACNoYDDAYSchJwAR6itFRYGJOmhYHhTEXiDKK1yM9ISzx0s00ifWZ778MHPxjs5Bl4zj7Aq8XgRtqBw7ltErmzeY4bG/YwJBs28LYb4Ncind4A1jJPIo1NgoeBOYGBn42Aw2ZDtKTLSaSx//zDUE9YC8NtoMNyzkgkSANtYeZhOJzAwNuGX4fB/WcJh3MqJAxn9hxjlpYxOG7YxnOMgMN6jhl/zjGok5c43sb48U1FtTw/TxoBh6FZComnUTAKRsEoGAUUAgAEgD3RwSTLEQAAAABJRU5ErkJggg==","orcid":"","institution":"Korea University","correspondingAuthor":true,"prefix":"","firstName":"Sungwoong","middleName":"","lastName":"Kim","suffix":""},{"id":630414606,"identity":"49718e79-eecd-417b-bb13-71008cc079a9","order_by":1,"name":"Chanhui Lee","email":"","orcid":"https://orcid.org/0009-0003-2092-3670","institution":"Korea University","correspondingAuthor":false,"prefix":"","firstName":"Chanhui","middleName":"","lastName":"Lee","suffix":""},{"id":630414607,"identity":"5c141884-3f86-431f-b2b8-7fd2b2ef1a80","order_by":2,"name":"Hanbum Ko","email":"","orcid":"","institution":"Korea University","correspondingAuthor":false,"prefix":"","firstName":"Hanbum","middleName":"","lastName":"Ko","suffix":""},{"id":630414608,"identity":"181b9484-ad06-41a6-a768-26102e4f313f","order_by":3,"name":"Yuheon Song","email":"","orcid":"https://orcid.org/0009-0003-7281-9285","institution":"UNIST","correspondingAuthor":false,"prefix":"","firstName":"Yuheon","middleName":"","lastName":"Song","suffix":""},{"id":630414609,"identity":"74c1ceaf-0f3e-4fb5-bef8-d1ff67db5c75","order_by":4,"name":"Yongjun Jeong","email":"","orcid":"","institution":"Korea University","correspondingAuthor":false,"prefix":"","firstName":"Yongjun","middleName":"","lastName":"Jeong","suffix":""},{"id":630414610,"identity":"83c90c1f-be3e-422a-bd3d-b6aa25499fda","order_by":5,"name":"Rodrigo Hormazabal","email":"","orcid":"","institution":"LG AI Research","correspondingAuthor":false,"prefix":"","firstName":"Rodrigo","middleName":"","lastName":"Hormazabal","suffix":""},{"id":630414611,"identity":"8837b287-3d67-4b21-928a-8a00ba822b25","order_by":6,"name":"Sehui Han","email":"","orcid":"","institution":"LG AI Research","correspondingAuthor":false,"prefix":"","firstName":"Sehui","middleName":"","lastName":"Han","suffix":""},{"id":630414612,"identity":"f90b8322-2458-434b-89da-dfc44cf11c21","order_by":7,"name":"Kyunghoon Bae","email":"","orcid":"","institution":"LG AI Research, Ministry of Science and ICT (MSIT)","correspondingAuthor":false,"prefix":"","firstName":"Kyunghoon","middleName":"","lastName":"Bae","suffix":""},{"id":630414613,"identity":"34d6c70c-83bc-4b23-b671-79202005cb06","order_by":8,"name":"Sungbin Lim","email":"","orcid":"","institution":"LG AI Research, Korea University","correspondingAuthor":false,"prefix":"","firstName":"Sungbin","middleName":"","lastName":"Lim","suffix":""}],"badges":[],"createdAt":"2025-11-25 07:15:43","currentVersionCode":1,"declarations":"","doi":"10.21203/rs.3.rs-8199982/v1","doiUrl":"https://doi.org/10.21203/rs.3.rs-8199982/v1","draftVersion":[],"editorialEvents":[],"editorialNote":"","failedWorkflow":false,"files":[{"id":108181488,"identity":"9e56fbb9-2bbd-44b8-b85c-343d91a669df","added_by":"auto","created_at":"2026-04-30 08:58:41","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":5945719,"visible":true,"origin":"","legend":"Article File","description":"","filename":"molllmmanuscriptrevised.pdf","url":"https://assets-eu.researchsquare.com/files/rs-8199982/v1_covered_930e44c1-72da-43c4-b43b-aa71d4e5529e.pdf"}],"financialInterests":"There is \u003cb\u003eNO\u003c/b\u003e Competing Interest.","formattedTitle":"Multimodal Molecular LLM with Improved Graph Utilization","fulltext":[],"fulltextSource":"","fullText":"","funders":[],"hasAdminPriorityOnWorkflow":false,"hasManuscriptDocX":false,"hasOptedInToPreprint":true,"hasPassedJournalQc":"","hasAnyPriority":true,"hideJournal":false,"highlight":"","institution":"","isAcceptedByJournal":false,"isAuthorSuppliedPdf":true,"isDeskRejected":"","isHiddenFromSearch":false,"isInQc":false,"isInWorkflow":false,"isPdf":true,"isPdfUpToDate":true,"isWithdrawnOrRetracted":false,"journal":{"display":true,"email":"
[email protected]","identity":"nature-portfolio","isNatureJournal":true,"hasQc":false,"allowDirectSubmit":false,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"","title":"Nature Portfolio","twitterHandle":"","acdcEnabled":false,"dfaEnabled":false,"editorialSystem":"ejp","reportingPortfolio":"","inReviewEnabled":true,"inReviewRevisionsEnabled":false},"keywords":"Molecular LLM, Multimodal LLM, Foundation model, Preference optimization, Generalization","lastPublishedDoi":"10.21203/rs.3.rs-8199982/v1","lastPublishedDoiUrl":"https://doi.org/10.21203/rs.3.rs-8199982/v1","license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"manuscriptAbstract":"Large language models (LLMs) are increasingly applied to molecular science, where instruction-tuning enables one model to address diverse tasks, complementing costly or narrow task-specific methods.\r\nYet most molecular LLMs operate on molecular sequences like SMILES, leaving molecular structure reasoning implicit.\r\nRecent works augment molecular sequences by concatenating multimodal representations like 2D graphs.\r\nHowever, we observe that naive instruction-tuning through next-token prediction causes models to bypass graph information, including bond connectivity and substructure, which provide important cues for solving diverse molecular tasks.\r\nTo resolve this, we introduce Molecular structure Preference Optimization (MolPO), a training objective where an LLM learns to generate answers by conditioning on correct graphs rather than perturbed ones.\r\nFurthermore, we introduce a hybrid graph encoder to enhance graph utilization, and leverage extensive instruction-tuning through staged training.\r\nThe resulting model, Mol-LLM, demonstrates strong performance across broader tasks than prior generalist molecular LLMs, including robust out-of-distribution generalization where they degrade.","manuscriptTitle":"Multimodal Molecular LLM with Improved Graph Utilization","msid":"","msnumber":"","nonDraftVersions":[{"code":1,"date":"2026-04-28 19:26:16","doi":"10.21203/rs.3.rs-8199982/v1","editorialEvents":[],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"nature-communications","isNatureJournal":true,"hasQc":false,"allowDirectSubmit":false,"externalIdentity":"NCOMMS","sideBox":"Learn more about [Nature Communications](http://www.nature.com/ncomms/)","snPcode":"","submissionUrl":"https://mts-ncomms.nature.com/","title":"Nature Communications","twitterHandle":"","acdcEnabled":true,"dfaEnabled":true,"editorialSystem":"ejp","reportingPortfolio":"Nature Communications","inReviewEnabled":true,"inReviewRevisionsEnabled":false}}],"origin":"","ownerIdentity":"7491c348-dc6d-48ff-9309-44d914aa1137","owner":[],"postedDate":"April 28th, 2026","published":true,"recentEditorialEvents":[{"type":"reviewerAgreed","content":"This content is not available.","date":"2026-05-12T16:03:28+00:00","index":3,"fulltext":"This content is not available."},{"type":"reviewerAgreed","content":"This content is not available.","date":"2026-05-11T11:33:31+00:00","index":2,"fulltext":"This content is not available."},{"type":"editorInvitedReview","content":"This content is not available.","date":"2026-05-09T02:58:50+00:00","index":1,"fulltext":"This content is not available."},{"type":"reviewerAgreed","content":"This content is not available.","date":"2026-05-09T02:29:41+00:00","index":1,"fulltext":"This content is not available."},{"type":"reviewersInvited","content":"5","date":"2026-05-08T22:27:18+00:00","index":"","fulltext":""}],"rejectedJournal":[],"revision":"","amendment":"","status":"under-review","subjectAreas":[{"id":67097675,"name":"Physical sciences/Chemistry"},{"id":67097676,"name":"Physical sciences/Chemistry/Chemical synthesis"},{"id":67097677,"name":"Physical sciences/Chemistry/Cheminformatics"},{"id":67097678,"name":"Physical sciences/Chemistry/Organic chemistry"},{"id":67097679,"name":"Physical sciences/Chemistry/Organic chemistry/Structure elucidation"}],"tags":[],"updatedAt":"2026-05-08T22:30:22+00:00","versionOfRecord":[],"versionCreatedAt":"2026-04-28 19:26:16","video":"","vorDoi":"","vorDoiUrl":"","workflowStages":[]},"version":"v1","identity":"rs-8199982","journalConfig":"researchsquare"},"__N_SSP":true},"page":"/article/[identity]/[[...version]]","query":{"redirect":"/article/rs-8199982","identity":"rs-8199982","version":["v1"]},"buildId":"XKTyCvWXoU3ODBz1xrDgd","isFallback":false,"isExperimentalCompile":false,"dynamicIds":[84888],"gssp":true,"scriptLoader":[]}
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.