Protein engineering in the big data era: harnessing near-redundant structural data | Research Square window.SnipcartSettings = { analytics: { enabled: false } }; (function() { var accessVector = localStorage.getItem('access_vector') || ''; window.dataLayer = window.dataLayer || []; if (accessVector) { window.dataLayer.push({ user: { profile: { profileInfo: { snid: accessVector } } } }); } })(); (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src='https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-K279D39R'); Browse Preprints In Review Journals COVID-19 Preprints AJE Video Bytes Research Tools Research Promotion AJE Professional Editing AJE Rubriq About Preprint Platform In Review Editorial Policies Our Team Advisory Board Help Center Sign In Submit a Preprint Cite Share Download PDF Research Article Protein engineering in the big data era: harnessing near-redundant structural data Samuel Coulbourn Flores, Athanasios Alexiou, Anastasios Glaros This is a preprint; it has not been peer reviewed by a journal. https://doi.org/ 10.21203/rs.3.rs-129805/v2 This work is licensed under a CC BY 4.0 License Status: Published Journal Publication published 01 Nov, 2021 Read the published version in PLOS ONE → Version 2 posted You are reading this latest preprint version Show more versions Abstract Motivation: Predicting the effect of mutations on protein-protein interactions is important for relating structure to function, as well as for in silico affinity maturation. The effect of mutations on protein-protein binding energy (ΔΔG) can be predicted by a variety of atomic simulation methods involving full or limited flexibility, and explicit or implicit solvent. Methods which consider only limited flexibility are naturally more economical, and many of them are quite accurate, however results are dependent on the atomic coordinate set used. In this work we perform a sequence and structure based search of the Protein Data Bank to find additional coordinate sets and repeat the calculation on each. Results: . We improve increase precision and Positive Predictive Value, and decrease Root Mean Square Error and higher Positive Predictive Value, compared to using single structures. Given the ongoing growth of near-redundant structures in the Protein Data Bank, our method will only increase in applicability and accuracy. Availability: Public web server at biodesign.scilifelab.se Bioinformatics Structural Biology Computational Biology Molecular Biology protein-protein interactions homology affinity complexes biologic design mutation protein engineering Figures Figure 1 Figure 2 Figure 3 Figure 4 Figure 5 Figure 6 Full Text Due to technical limitations, full-text HTML conversion of this manuscript could not be completed. However, the latest manuscript can be downloaded and accessed as a PDF. Supplementary Files supplementardata.pdf Cite Share Download PDF Status: Published Journal Publication published 01 Nov, 2021 Read the published version in PLOS ONE → Version 2 posted You are reading this latest preprint version Show more versions Research Square lets you share your work early, gain feedback from the community, and start making changes to your manuscript prior to peer review in a journal. As a division of Research Square Company, we’re committed to making research communication faster, fairer, and more useful. We do this by developing innovative software and high quality services for the global research community. Our growing team is made up of researchers and industry professionals working together to solve the most critical problems facing scientific publishing. Also discoverable on Platform About Our Team In Review Editorial Policies Advisory Board Help Center Resources Author Services Accessibility API Access RSS feed Manage Cookie Preferences © Research Square 2026 | ISSN 2693-5015 (online) Privacy Policy Terms of Service Do Not Sell My Personal Information {"props":{"pageProps":{"initialData":{"identity":"rs-129805","acceptedTermsAndConditions":true,"allowDirectSubmit":true,"archivedVersions":[],"articleType":"Research Article","associatedPublications":[],"authors":[{"id":6609208,"identity":"b5a9ee86-8447-41e8-9949-d71a77b21dca","order_by":0,"name":"Samuel Coulbourn Flores","email":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAZAAAAAyAQMAAABI0h/eAAAABlBMVEX///8AAABVwtN+AAAACXBIWXMAAA7EAAAOxAGVKw4bAAABHElEQVRIie2QsWrDMBRFnxDIixOtCh78C69kCpR+S0zBUx0KWQwtrkzAXlK6eij9h1LwHDAkQwNdPdZ0DribDRkixx1V07GDziDERffddwVgMPxbboGPz5cQgFlE1kqZD1sQJsn5slcWSuJMKX+0kO6kQAYtbvruixpBMOuxqtqXyB1TGssarxbAi412/j7IJ1lnsXfT6SgvLhK1WJzh9RKEr01CCHLHRoiY8JlD8s1cdUm/bKSeFErXLfZ0yJ1jlyJ8q22fo84Sr4744En+UWvLlCoFeguDkaS9BbDwJNzo/6o8vM3WKFSXLXXs7U+XNe6WasgviwWvZRNeCm4l5Lu5j1w3TSvZhHcLzotPbUyP0Ghs4L3BYDAYhjkB6WJUsU3jT5AAAAAASUVORK5CYII=","orcid":"https://orcid.org/0000-0002-3869-8147","institution":"Stockholm University","correspondingAuthor":true,"submittingAuthor":false,"prefix":"","firstName":"Samuel","middleName":"Coulbourn","lastName":"Flores","suffix":""},{"id":6609209,"identity":"24e15b00-78a7-4498-aaa3-3125558c4a53","order_by":1,"name":"Athanasios Alexiou","email":"","orcid":"","institution":"University of Thessaly","correspondingAuthor":false,"submittingAuthor":false,"prefix":"","firstName":"Athanasios","middleName":"","lastName":"Alexiou","suffix":""},{"id":6609210,"identity":"7275db49-eb8c-450b-9af4-c3cf9c74084f","order_by":2,"name":"Anastasios Glaros","email":"","orcid":"","institution":"Science For Life Laboratory","correspondingAuthor":false,"submittingAuthor":false,"prefix":"","firstName":"Anastasios","middleName":"","lastName":"Glaros","suffix":""}],"badges":[],"createdAt":"2020-12-16 11:38:32","currentVersionCode":2,"declarations":{"humanSubjects":false,"vertebrateSubjects":false,"conflictsOfInterestStatement":true,"humanSubjectEthicalGuidelines":false,"humanSubjectConsent":false,"humanSubjectClinicalTrial":false,"humanSubjectCaseReport":false,"vertebrateSubjectEthicalGuidelines":false,"coiExplicitlySet":false},"doi":"10.21203/rs.3.rs-129805/v2","doiUrl":"https://doi.org/10.21203/rs.3.rs-129805/v2","draftVersion":[],"editorialEvents":[{"content":"https://doi.org/10.1371/journal.pone.0257614","type":"published","date":"2021-11-02T00:00:00+00:00"}],"editorialNote":"","failedWorkflow":false,"files":[{"id":6103911,"identity":"18360a55-8b5e-41e9-9bb2-bb622d282315","added_by":"auto","created_at":"2021-02-18 20:00:06","extension":"png","order_by":1,"title":"Figure 1","display":"","copyAsset":false,"role":"figure","size":47893,"visible":true,"origin":"","legend":"Program flow. The user must provide an initial Protein Data Bank (PDB) ID, specify which relevant chains are in which of two interacting complexes (irrelevant chains may be left out). 1. The fasta_lwp program searches the PDB for structures con-taining chains homologous (E-value below eValueCutoff, here 10-11) to those specified by the user. 2. We group the thus-discovered homolog chains by PDB ID, each such PDB ID is re-ferred to as a “homolog.” We loop over the homologs, per-forming three checks on each. 3. As a first check, we determine whether the thus-discovered homologs contain chains corre-sponding to all those specified by the user; those not having all such chains are discarded. 4. Homologs in which all chains do not have at least 90% sequence identity vs. the corresponding user-specified chain are discarded. 5. We perform a rigid alignment of the entire homolog against the user-specified structure, based only on the user-specified chains. Non-corresponding (extraneous) chains are moved along with the rest of the complex. This is the most computationally-expensive process, but only needs to be done once for homo-log that makes it to this step; results of all three checks are saved persistently. 6. If RMSD \u003e 6.0 Å (again based on corre-sponding chains), we discard the homolog. Most homologs which are rejected at this step contain the correct chains but in a different configuration. 7. We then compute the ΔΔG for the user-requested mutation, using the homolog structure and FoldX4. Steps 3-7 are repeated for each homolog. 8. We aver-age ΔΔG over all homologs that reached and completed step 7 and report the result. ","description":"","filename":"1.png","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2/464aa63d218da2d849be6c3a.png"},{"id":6103949,"identity":"ba8af676-a46a-46d8-97c2-691e04f0e147","added_by":"auto","created_at":"2021-02-18 20:03:07","extension":"png","order_by":2,"title":"Figure 2","display":"","copyAsset":false,"role":"figure","size":407199,"visible":true,"origin":"","legend":"Illustration of the sequence and structure matching proce-dure.","description":"","filename":"2.png","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2/a0acbd322c12db63f01fa20b.png"},{"id":6103916,"identity":"0a823748-8dcf-47fa-a519-d2b79befca71","added_by":"auto","created_at":"2021-02-18 20:00:07","extension":"png","order_by":3,"title":"Figure 3","display":"","copyAsset":false,"role":"figure","size":80471,"visible":true,"origin":"","legend":"Scatterplot of ΔΔGpredicted vs. ΔΔGexperimental, for Dataset C (sin-gle-position substitutions, where more than one structure was available). Green circles: mutants averaged over multiple structures, N=522. Black dots: mutants computed on a single structure -- as multiple structures were available for each mutant, this has a higher N=4028. Note the clear outliers are all single-structure points. Note the third quadrant is populated with True Positives -- ΔΔGpredicted and ΔΔGexperimental both negative. On the other hand, the fourth quadrant, representing False Positives, does not have any multiple-structure results below ΔΔGpredicted \u003c -0.65 kcal/mol. The improvement in Positive Predictive Value is discussed elsewhere in this work.","description":"","filename":"3.png","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2/9b9b016aa992f8d3e0efee06.png"},{"id":6103912,"identity":"d69fd64e-f682-4142-a817-a920e8f7d6f1","added_by":"auto","created_at":"2021-02-18 20:00:07","extension":"png","order_by":4,"title":"Figure 4","display":"","copyAsset":false,"role":"figure","size":80690,"visible":true,"origin":"","legend":"Receiver Operating Characteristic, comparing homologyScan-ner vs. calculation on single structures.","description":"","filename":"4.png","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2/ce543a0190391f07f99c2576.png"},{"id":6103950,"identity":"82ebb2b2-4562-470e-8443-2d4813021082","added_by":"auto","created_at":"2021-02-18 20:03:07","extension":"png","order_by":5,"title":"Figure 5","display":"","copyAsset":false,"role":"figure","size":153587,"visible":true,"origin":"","legend":"Positive Predictive Value (PPV) for single vs. multiple struc-tures. TP + FP is the denominator of PPV, so we emphasize that this quantity becomes small for ΔΔGpredicted \u003c -1 kcal/mol (crosses). This is why the PPV becomes erratic, at least for sin-gle structures.","description":"","filename":"5.png","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2/20993e312ca66d77c3d233a2.png"},{"id":6103947,"identity":"e16d5c76-f9ec-4a8c-8627-70a8ad277c66","added_by":"auto","created_at":"2021-02-18 20:03:07","extension":"png","order_by":6,"title":"Figure 6","display":"","copyAsset":false,"role":"figure","size":591193,"visible":true,"origin":"","legend":"The homologyScanner public web server. Users can provide PDB ID, chose chains in each of two subunits, and specify a mu-tation to be computed. FoldX ΔΔG is computed for the query and all matching complexes and reported to the user. The re-sults are available for browsing by others. Compute nodes are needed only for high-throughput runs. The software compo-nents are available on github, simtk.org, and dockerhub. A server has also been set up on a single-board computer for private deployment.","description":"","filename":"6.png","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2/fe51f5c49e774a3074b650ae.png"},{"id":16437637,"identity":"d1d3c1e2-dcb9-4ac3-96b2-278994cafbe0","added_by":"auto","created_at":"2021-12-14 13:53:58","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":1277829,"visible":true,"origin":"","legend":"","description":"","filename":"HomologyScannerMS.7.ResearchSquare1.pdf","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2_covered.pdf"},{"id":13590768,"identity":"4f808098-c012-421d-bf7f-c0bc8e3e14b5","added_by":"auto","created_at":"2021-09-17 05:06:07","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":1272648,"visible":true,"origin":"","legend":"","description":"","filename":"HomologyScannerMS.7.ResearchSquare1.pdf","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2_covered.pdf"},{"id":13565734,"identity":"05db89e4-bb7e-4ebd-96f5-5e27ab1ea6d1","added_by":"auto","created_at":"2021-09-17 03:26:20","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":1400835,"visible":true,"origin":"","legend":"","description":"","filename":"HomologyScannerMS.6.ResearchSquare.pdf","url":"https://assets-eu.researchsquare.com/files/rs-129805/v1_covered.pdf"},{"id":6104077,"identity":"8179c890-af04-4bda-9209-20b2d7eb6594","added_by":"auto","created_at":"2021-02-18 20:09:11","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":1008014,"visible":true,"origin":"","legend":"","description":"","filename":"HomologyScannerMS.7.ResearchSquare1.pdf","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2_stamped.pdf"},{"id":6104045,"identity":"7e647228-bfd6-4093-af69-928b6fe52099","added_by":"auto","created_at":"2021-02-18 20:06:07","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"supplement","size":98276,"visible":true,"origin":"","legend":"","description":"","filename":"supplementardata.pdf","url":"https://assets-eu.researchsquare.com/files/rs-129805/v2/3fe8b67cb61e490fa24a9e2c.pdf"}],"financialInterests":"","formattedTitle":"\u003cp\u003eProtein engineering in the big data era: harnessing near-redundant structural data\u003c/p\u003e\u003cp\u003e\u003cbr\u003e\u003c/p\u003e","fulltext":[{"header":"Full Text","content":"Due to technical limitations, full-text HTML conversion of this manuscript could not be completed. However, the latest manuscript can be downloaded and \u003ca href='/article/rs-129805/latest.pdf' target='_blank'\u003e accessed as a PDF.\u003c/a\u003e"}],"fulltextSource":"","fullText":"","funders":[],"hasAdminPriorityOnWorkflow":false,"hasManuscriptDocX":false,"hasOptedInToPreprint":true,"hasPassedJournalQc":"","hasAnyPriority":true,"hideJournal":false,"highlight":"","institution":"","isAcceptedByJournal":true,"isAuthorSuppliedPdf":true,"isDeskRejected":"","isHiddenFromSearch":false,"isInQc":false,"isInWorkflow":false,"isPdf":false,"isPdfUpToDate":true,"isWithdrawnOrRetracted":false,"journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true},"keywords":"protein-protein interactions, homology, affinity, complexes, biologic design, mutation, protein engineering","lastPublishedDoi":"10.21203/rs.3.rs-129805/v2","lastPublishedDoiUrl":"https://doi.org/10.21203/rs.3.rs-129805/v2","license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"manuscriptAbstract":"\u003cp\u003e\u003cstrong\u003eMotivation:\u003c/strong\u003e Predicting the effect of mutations on protein-protein interactions is important for relating structure to function, as well as for \u003cem\u003ein silico\u003c/em\u003e affinity maturation. The effect of mutations on protein-protein binding energy (ΔΔG) can be predicted by a variety of atomic simulation methods involving full or limited flexibility, and explicit or implicit solvent. Methods which consider only limited flexibility are naturally more economical, and many of them are quite accurate, however results are dependent on the atomic coordinate set used. In this work we perform a sequence and structure based search of the Protein Data Bank to find additional coordinate sets and repeat the calculation on each.\u003c/p\u003e\u003cp\u003e\u003cstrong\u003eResults:\u003c/strong\u003e . We improve increase precision and Positive Predictive Value, and decrease Root Mean Square Error and higher Positive Predictive Value, compared to using single structures. Given the ongoing growth of near-redundant structures in the Protein Data Bank, our method will only increase in applicability and accuracy.\u003c/p\u003e\u003cp\u003e\u003cstrong\u003eAvailability:\u003c/strong\u003e Public web server at biodesign.scilifelab.se\u0026nbsp;\u003c/p\u003e","manuscriptTitle":"Protein engineering in the big data era: harnessing near-redundant structural data","msid":"","msnumber":"","nonDraftVersions":[{"code":2,"date":"2021-02-18 20:00:05","doi":"10.21203/rs.3.rs-129805/v2","editorialEvents":[{"type":"communityComments","content":0}],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true}},{"code":1,"date":"2020-12-16 19:43:12","doi":"10.21203/rs.3.rs-129805/v1","editorialEvents":[{"type":"communityComments","content":0}],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true}}],"origin":"","ownerIdentity":"968a7c72-4adc-4f48-9c36-374502682cac","owner":[],"postedDate":"February 18th, 2021","published":true,"recentEditorialEvents":[],"rejectedJournal":[],"revision":"","amendment":"","status":"published-in-journal","subjectAreas":[{"id":1510941,"name":"Bioinformatics"},{"id":1510942,"name":"Structural Biology"},{"id":1510943,"name":"Computational Biology"},{"id":1510944,"name":"Molecular Biology"}],"tags":[],"updatedAt":"2021-12-14T13:53:44+00:00","versionOfRecord":{"articleIdentity":"rs-129805","link":"https://doi.org/10.1371/journal.pone.0257614","journal":{"identity":"plos-one","isVorOnly":true,"title":"PLOS ONE"},"publishedOn":"2021-11-02 00:00:00","publishedOnDateReadable":"November 2nd, 2021"},"versionCreatedAt":"2021-02-18 20:00:05","video":"","vorDoi":"10.1371/journal.pone.0257614","vorDoiUrl":"https://doi.org/10.1371/journal.pone.0257614","workflowStages":[]},"version":"v2","identity":"rs-129805","journalConfig":"researchsquare"},"__N_SSP":true},"page":"/article/[identity]/[[...version]]","query":{"redirect":"/article/rs-129805","identity":"rs-129805","version":["v2"]},"buildId":"WrCJVZZCHTDjtuVLN7oU0","isFallback":false,"isExperimentalCompile":false,"dynamicIds":[84888],"gssp":true,"scriptLoader":[]}
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.