Clustering and Classification for Dry Bean Feature Imbalanced Data | Research Square window.SnipcartSettings = { analytics: { enabled: false } }; (function() { var accessVector = localStorage.getItem('access_vector') || ''; window.dataLayer = window.dataLayer || []; if (accessVector) { window.dataLayer.push({ user: { profile: { profileInfo: { snid: accessVector } } } }); } })(); (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src='https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-K279D39R'); Browse Preprints In Review Journals COVID-19 Preprints AJE Video Bytes Research Tools Research Promotion AJE Professional Editing AJE Rubriq About Preprint Platform In Review Editorial Policies Our Team Advisory Board Help Center Sign In Submit a Preprint Cite Share Download PDF Article Clustering and Classification for Dry Bean Feature Imbalanced Data Chou-Yuan Lee, Wei Wang, Jian-Qiong Huang This is a preprint; it has not been peer reviewed by a journal. https://doi.org/ 10.21203/rs.3.rs-4201421/v1 This work is licensed under a CC BY 4.0 License Status: Published Journal Publication published 28 Dec, 2024 Read the published version in Scientific Reports → Version 1 posted 10 You are reading this latest preprint version Abstract The dry bean dataset of this study uses the imbalanced data set of the University of California, Irvine (UCI) machine learning warehouse platform. The dataset consists of 13,611 data points, including 16 explanatory variables and 1 target variable (Class). This paper proposes clustering and classification of dry beans based on K-means combined with machine learning methods such as decision tree (DT), random forest (RF) and support vector machine (SVM). The key idea is that K-means performs clustering first, and then imports the clustering results into DT, RF, and SVM to obtain the classification accuracy and compares it with traditional machine learning methods to improve the classification accuracy of dry bean imbalanced data. The experimental result is shown that the proposed method found K-means + SVM to have good classification accuracy on the dry bean dataset. This study further explored the 11 decision rules obtained in K-means + DT. In the ranking of explanatory variable importance in K-means + RF, it was found that the top three were Compactness, ShapeFactor3, and ShapeFactor1. Therefore, the proposed method of first clustering and then classifying can be used as an effective means to analyze dry bean imbalanced data. Biological sciences/Plant sciences Physical sciences/Mathematics and computing decision tree random forest support vector machine imbalanced data Full Text Additional Declarations No competing interests reported. Cite Share Download PDF Status: Published Journal Publication published 28 Dec, 2024 Read the published version in Scientific Reports → Version 1 posted Editorial decision: Revision requested 24 May, 2024 Reviews received at journal 09 May, 2024 Reviewers agreed at journal 06 May, 2024 Reviews received at journal 01 May, 2024 Reviewers agreed at journal 15 Apr, 2024 Reviewers invited by journal 09 Apr, 2024 Editor assigned by journal 09 Apr, 2024 Editor invited by journal 09 Apr, 2024 Submission checks completed at journal 09 Apr, 2024 First submitted to journal 01 Apr, 2024 You are reading this latest preprint version Research Square lets you share your work early, gain feedback from the community, and start making changes to your manuscript prior to peer review in a journal. As a division of Research Square Company, we’re committed to making research communication faster, fairer, and more useful. We do this by developing innovative software and high quality services for the global research community. Our growing team is made up of researchers and industry professionals working together to solve the most critical problems facing scientific publishing. Also discoverable on Platform About Our Team In Review Editorial Policies Advisory Board Help Center Resources Author Services Accessibility API Access RSS feed Manage Cookie Preferences © Research Square 2026 | ISSN 2693-5015 (online) Privacy Policy Terms of Service Do Not Sell My Personal Information {"props":{"pageProps":{"initialData":{"identity":"rs-4201421","acceptedTermsAndConditions":true,"allowDirectSubmit":false,"archivedVersions":[],"articleType":"Article","associatedPublications":[],"authors":[{"id":290298870,"identity":"7dec43da-93f8-409c-97eb-945d78ee8106","order_by":0,"name":"Chou-Yuan Lee","email":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAZAAAAAyAQMAAABI0h/eAAAABlBMVEX///8AAABVwtN+AAAACXBIWXMAAA7EAAAOxAGVKw4bAAAAtElEQVRIiWNgGAWjYBACPgYGxgcVDHIgtgFxWtgYGJgNzjAYk6aFTYJELew5ZhUH/hgkNrA3b5NgqLlDhBaeN2Y3DvAAtfAcK5NgOPaMCC0SOWa3P0j8SWwAMiQYGw4Tp6XggAHQFvk3JGhhOJAA1CLBQ6wWnmfFEgcOGBi38aQVWyQcI0ILP3vyxg/AEJPtZz+88caHGiK0MDBkQKKDDUQkEKOBgSH9AXHqRsEoGAWjYOQCAJ8DNYAFxrUpAAAAAElFTkSuQmCC","orcid":"","institution":"Fuzhou University of International Studies and Trade","correspondingAuthor":true,"submittingAuthor":false,"prefix":"","firstName":"Chou-Yuan","middleName":"","lastName":"Lee","suffix":""},{"id":290298871,"identity":"6101b084-58c8-428c-a0c9-39b118b10f38","order_by":1,"name":"Wei Wang","email":"","orcid":"","institution":"Yunnan University","correspondingAuthor":false,"submittingAuthor":false,"prefix":"","firstName":"Wei","middleName":"","lastName":"Wang","suffix":""},{"id":290298873,"identity":"92ae5698-f2e0-4635-b5e8-12756f30b407","order_by":2,"name":"Jian-Qiong Huang","email":"","orcid":"","institution":"Fuzhou University of International Studies and Trade","correspondingAuthor":false,"submittingAuthor":false,"prefix":"","firstName":"Jian-Qiong","middleName":"","lastName":"Huang","suffix":""}],"badges":[],"createdAt":"2024-04-01 14:59:14","currentVersionCode":1,"declarations":"","doi":"10.21203/rs.3.rs-4201421/v1","doiUrl":"https://doi.org/10.21203/rs.3.rs-4201421/v1","draftVersion":[],"editorialEvents":[{"content":"https://doi.org/10.1038/s41598-024-82253-6","type":"published","date":"2024-12-28T15:57:04+00:00"}],"editorialNote":"","failedWorkflow":false,"files":[{"id":72640402,"identity":"4b41fe49-6d91-48a1-b215-5f6a1ee7fc92","added_by":"auto","created_at":"2024-12-30 16:05:52","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":582539,"visible":true,"origin":"","legend":"","description":"","filename":"Clusteringandclassificationfordrybeanfeatureimbalanceddata.pdf","url":"https://assets-eu.researchsquare.com/files/rs-4201421/v1_covered_fc1ff457-42b3-4db6-afd8-a425c7e1f409.pdf"}],"financialInterests":"No competing interests reported.","formattedTitle":"Clustering and Classification for Dry Bean Feature Imbalanced Data","fulltext":[],"fulltextSource":"","fullText":"","funders":[],"hasAdminPriorityOnWorkflow":false,"hasManuscriptDocX":false,"hasOptedInToPreprint":true,"hasPassedJournalQc":"","hasAnyPriority":false,"hideJournal":false,"highlight":"","institution":"","isAcceptedByJournal":true,"isAuthorSuppliedPdf":true,"isDeskRejected":"","isHiddenFromSearch":false,"isInQc":false,"isInWorkflow":false,"isPdf":true,"isPdfUpToDate":true,"isWithdrawnOrRetracted":false,"journal":{"display":true,"email":"
[email protected]","identity":"scientific-reports","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":false,"externalIdentity":"scirep","sideBox":"Learn more about [Scientific Reports](http://www.nature.com/srep/)","snPcode":"","submissionUrl":"","title":"Scientific Reports","twitterHandle":"","acdcEnabled":true,"dfaEnabled":true,"editorialSystem":"stoa","reportingPortfolio":"Scientific Reports","inReviewEnabled":true,"inReviewRevisionsEnabled":true},"keywords":"decision tree, random forest, support vector machine, imbalanced data","lastPublishedDoi":"10.21203/rs.3.rs-4201421/v1","lastPublishedDoiUrl":"https://doi.org/10.21203/rs.3.rs-4201421/v1","license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"manuscriptAbstract":"The dry bean dataset of this study uses the imbalanced data set of the University of California, Irvine (UCI) machine learning warehouse platform. The dataset consists of 13,611 data points, including 16 explanatory variables and 1 target variable (Class). This paper proposes clustering and classification of dry beans based on K-means combined with machine learning methods such as decision tree (DT), random forest (RF) and support vector machine (SVM). The key idea is that K-means performs clustering first, and then imports the clustering results into DT, RF, and SVM to obtain the classification accuracy and compares it with traditional machine learning methods to improve the classification accuracy of dry bean imbalanced data. The experimental result is shown that the proposed method found K-means + SVM to have good classification accuracy on the dry bean dataset. This study further explored the 11 decision rules obtained in K-means + DT. In the ranking of explanatory variable importance in K-means + RF, it was found that the top three were Compactness, ShapeFactor3, and ShapeFactor1. Therefore, the proposed method of first clustering and then classifying can be used as an effective means to analyze dry bean imbalanced data.","manuscriptTitle":"Clustering and Classification for Dry Bean Feature Imbalanced Data","msid":"","msnumber":"","nonDraftVersions":[{"code":1,"date":"2024-04-12 04:57:31","doi":"10.21203/rs.3.rs-4201421/v1","editorialEvents":[{"type":"communityComments","content":0},{"type":"decision","content":"Revision requested","date":"2024-05-24T07:58:18+00:00","index":"","fulltext":""},{"type":"editorInvitedReview","content":"","date":"2024-05-10T01:07:03+00:00","index":"hide","fulltext":""},{"type":"reviewerAgreed","content":"098c730f-3ec4-4dbf-82c0-4c6da9fca8a2","date":"2024-05-06T14:28:51+00:00","index":"hide","fulltext":""},{"type":"editorInvitedReview","content":"","date":"2024-05-01T11:57:43+00:00","index":"hide","fulltext":""},{"type":"reviewerAgreed","content":"c1a39754-0747-490c-b7f6-43a217c7446a","date":"2024-04-15T11:11:50+00:00","index":"hide","fulltext":""},{"type":"reviewersInvited","content":"","date":"2024-04-10T01:09:58+00:00","index":"","fulltext":""},{"type":"editorAssigned","content":"","date":"2024-04-10T01:07:52+00:00","index":"","fulltext":""},{"type":"editorInvited","content":"","date":"2024-04-09T18:48:36+00:00","index":"","fulltext":""},{"type":"checksComplete","content":"","date":"2024-04-09T18:02:19+00:00","index":"","fulltext":""},{"type":"submitted","content":"Scientific Reports","date":"2024-04-01T14:55:27+00:00","index":"","fulltext":""}],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"scientific-reports","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":false,"externalIdentity":"scirep","sideBox":"Learn more about [Scientific Reports](http://www.nature.com/srep/)","snPcode":"","submissionUrl":"","title":"Scientific Reports","twitterHandle":"","acdcEnabled":true,"dfaEnabled":true,"editorialSystem":"stoa","reportingPortfolio":"Scientific Reports","inReviewEnabled":true,"inReviewRevisionsEnabled":true}}],"origin":"","ownerIdentity":"9aaa9016-d196-4364-95c3-8339078600d5","owner":[],"postedDate":"April 12th, 2024","published":true,"recentEditorialEvents":[],"rejectedJournal":[],"revision":"","amendment":"","status":"published-in-journal","subjectAreas":[{"id":30577975,"name":"Biological sciences/Plant sciences"},{"id":30577976,"name":"Physical sciences/Mathematics and computing"}],"tags":[],"updatedAt":"2024-12-30T15:59:10+00:00","versionOfRecord":{"articleIdentity":"rs-4201421","link":"https://doi.org/10.1038/s41598-024-82253-6","journal":{"identity":"scientific-reports","isVorOnly":false,"title":"Scientific Reports"},"publishedOn":"2024-12-28 15:57:04","publishedOnDateReadable":"December 28th, 2024"},"versionCreatedAt":"2024-04-12 04:57:31","video":"","vorDoi":"10.1038/s41598-024-82253-6","vorDoiUrl":"https://doi.org/10.1038/s41598-024-82253-6","workflowStages":[]},"version":"v1","identity":"rs-4201421","journalConfig":"researchsquare"},"__N_SSP":true},"page":"/article/[identity]/[[...version]]","query":{"redirect":"/article/rs-4201421","identity":"rs-4201421","version":["v1"]},"buildId":"ApUGefWb6u5IBVtyqm6d5","isFallback":false,"isExperimentalCompile":false,"dynamicIds":[84888],"gssp":true,"scriptLoader":[]}
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.