Intelligent Detection of Mobile SMS Spam via Machine Learning and Deep Learning | Research Square window.SnipcartSettings = { analytics: { enabled: false } }; (function() { var accessVector = localStorage.getItem('access_vector') || ''; window.dataLayer = window.dataLayer || []; if (accessVector) { window.dataLayer.push({ user: { profile: { profileInfo: { snid: accessVector } } } }); } })(); (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src='https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-K279D39R'); Browse Preprints In Review Journals COVID-19 Preprints AJE Video Bytes Research Tools Research Promotion AJE Professional Editing AJE Rubriq About Preprint Platform In Review Editorial Policies Our Team Advisory Board Help Center Sign In Submit a Preprint Cite Share Download PDF Research Article Intelligent Detection of Mobile SMS Spam via Machine Learning and Deep Learning Megha Birthare, Neelesh Jain This is a preprint; it has not been peer reviewed by a journal. https://doi.org/ 10.21203/rs.3.rs-6842737/v1 This work is licensed under a CC BY 4.0 License Status: Posted Version 1 posted You are reading this latest preprint version Abstract The global rise in social media usage has led to a surge in unwanted bulk SMS, necessitating the development of an effective system to filter out these messages. The most prevalent issue on the internet is spam text messages. Sending a spam-filled SMS is a straightforward task for spammers. Spammers are able to take valuable data, including contacts and files, from our devices. In recent years, several word embedding techniques leveraging deep learning have been developed. These advancements in word representation could offer a reliable remedy for these problems This study will look at a technique that employs natural language processing to distinguish among spam and ham texts utilizing the SMS Spam Collection Dataset from the UCI Machine Learning Repository. We compared the accuracy and outcomes of using the Bi-LSTM and LSTM. The effectiveness of the dataset is assessed using measures like F1-score, recall, and accuracy. The study demonstrates that the dataset's overall accuracy increases when Bi-LSTM classification is used. Python is used for all work, and a Jupyter notebook is used for implementation. SMS Spam Classification Bidirectional LSTM (Bi-LSTM) Long Short-Term Memory Networks (LSTM) Full Text Additional Declarations No competing interests reported. Cite Share Download PDF Status: Posted Version 1 posted You are reading this latest preprint version Research Square lets you share your work early, gain feedback from the community, and start making changes to your manuscript prior to peer review in a journal. As a division of Research Square Company, we’re committed to making research communication faster, fairer, and more useful. We do this by developing innovative software and high quality services for the global research community. Our growing team is made up of researchers and industry professionals working together to solve the most critical problems facing scientific publishing. Also discoverable on Platform About Our Team In Review Editorial Policies Advisory Board Help Center Resources Author Services Accessibility API Access RSS feed Manage Cookie Preferences © Research Square 2026 | ISSN 2693-5015 (online) Privacy Policy Terms of Service Do Not Sell My Personal Information {"props":{"pageProps":{"initialData":{"identity":"rs-6842737","acceptedTermsAndConditions":true,"allowDirectSubmit":true,"archivedVersions":[],"articleType":"Research Article","associatedPublications":[],"authors":[{"id":468416707,"identity":"5c01ad49-d18f-4a83-96d5-40c7e33ea503","order_by":0,"name":"Megha Birthare","email":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAZAAAAAyAQMAAABI0h/eAAAABlBMVEX///8AAABVwtN+AAAACXBIWXMAAA7EAAAOxAGVKw4bAAAA+klEQVRIiWNgGAWjYHACNgbGBgkGBgkQuwKImZkbiNPCA9ZyBqSFkSgtDBAtjG0gAQJadNvbrz34ucMi3166+dmHj/Nqo/nbgVp+VGzDqcXszJlyw94zEpY9MseMZ87cdjx3xmHGBsaeM7dxa7mRkybB2yZhwCORYMzMu+1YbgNQCzNjGx4t99+kSf4Fa0n/zPx3zrHc+QS13GA/Jg2xJccYGFY1uRsIajmTwyYtewao5UZOMWPPsQO5G4FaDuL1y/HjzyTf7qgzYJ+RvpnhR01d7rzzhw8++FGBWwswQgyQeYfB5AE86oGA/QEyrw6/4lEwCkbBKBiRAACprltJ6CalUgAAAABJRU5ErkJggg==","orcid":"","institution":"SAM Global University","correspondingAuthor":true,"prefix":"","firstName":"Megha","middleName":"","lastName":"Birthare","suffix":""},{"id":468416708,"identity":"cf0d8c5e-0b82-49d2-8f85-5f04c7f13082","order_by":1,"name":"Neelesh Jain","email":"","orcid":"","institution":"SAM Global University","correspondingAuthor":false,"prefix":"","firstName":"Neelesh","middleName":"","lastName":"Jain","suffix":""}],"badges":[],"createdAt":"2025-06-07 12:08:18","currentVersionCode":1,"declarations":"","doi":"10.21203/rs.3.rs-6842737/v1","doiUrl":"https://doi.org/10.21203/rs.3.rs-6842737/v1","draftVersion":[],"editorialEvents":[],"editorialNote":"","failedWorkflow":false,"files":[{"id":85155256,"identity":"eb4eb6a3-e2b2-44ca-a4d5-8d14f842431d","added_by":"auto","created_at":"2025-06-22 19:01:30","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":525321,"visible":true,"origin":"","legend":"","description":"","filename":"IntelligentDetectionofMobileSMSSpamviaMachineLearningandDeepLearning.pdf","url":"https://assets-eu.researchsquare.com/files/rs-6842737/v1_covered_e028f200-710a-4811-a8a6-be839a403131.pdf"}],"financialInterests":"No competing interests reported.","formattedTitle":"Intelligent Detection of Mobile SMS Spam via Machine Learning and Deep Learning","fulltext":[],"fulltextSource":"","fullText":"","funders":[],"hasAdminPriorityOnWorkflow":false,"hasManuscriptDocX":false,"hasOptedInToPreprint":true,"hasPassedJournalQc":"","hasAnyPriority":true,"hideJournal":true,"highlight":"","institution":"","isAcceptedByJournal":false,"isAuthorSuppliedPdf":true,"isDeskRejected":"","isHiddenFromSearch":false,"isInQc":false,"isInWorkflow":false,"isPdf":true,"isPdfUpToDate":true,"isWithdrawnOrRetracted":false,"journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true},"keywords":"SMS Spam Classification, Bidirectional LSTM (Bi-LSTM), Long Short-Term Memory Networks (LSTM)","lastPublishedDoi":"10.21203/rs.3.rs-6842737/v1","lastPublishedDoiUrl":"https://doi.org/10.21203/rs.3.rs-6842737/v1","license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"manuscriptAbstract":"\u003cp\u003eThe global rise in social media usage has led to a surge in unwanted bulk SMS, necessitating the development of an effective system to filter out these messages. The most prevalent issue on the internet is spam text messages. Sending a spam-filled SMS is a straightforward task for spammers. Spammers are able to take valuable data, including contacts and files, from our devices. In recent years, several word embedding techniques leveraging deep learning have been developed. These advancements in word representation could offer a reliable remedy for these problems This study will look at a technique that employs natural language processing to distinguish among spam and ham texts utilizing the SMS Spam Collection Dataset from the UCI Machine Learning Repository.\u003c/p\u003e \u003cp\u003eWe compared the accuracy and outcomes of using the Bi-LSTM and LSTM. The effectiveness of the dataset is assessed using measures like F1-score, recall, and accuracy. The study demonstrates that the dataset's overall accuracy increases when Bi-LSTM classification is used. Python is used for all work, and a Jupyter notebook is used for implementation.\u003c/p\u003e","manuscriptTitle":"Intelligent Detection of Mobile SMS Spam via Machine Learning and Deep Learning","msid":"","msnumber":"","nonDraftVersions":[{"code":1,"date":"2025-06-11 03:24:28","doi":"10.21203/rs.3.rs-6842737/v1","editorialEvents":[{"type":"communityComments","content":0}],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true}}],"origin":"","ownerIdentity":"2e893981-b9b5-44d3-855d-9114ecae31b7","owner":[],"postedDate":"June 11th, 2025","published":true,"recentEditorialEvents":[],"rejectedJournal":[],"revision":"","amendment":"","status":"posted","subjectAreas":[],"tags":[],"updatedAt":"2025-06-22T18:53:24+00:00","versionOfRecord":[],"versionCreatedAt":"2025-06-11 03:24:28","video":"","vorDoi":"","vorDoiUrl":"","workflowStages":[]},"version":"v1","identity":"rs-6842737","journalConfig":"researchsquare"},"__N_SSP":true},"page":"/article/[identity]/[[...version]]","query":{"redirect":"/article/rs-6842737","identity":"rs-6842737","version":["v1"]},"buildId":"8U1c8b4HqxoKbykW_rLl7","isFallback":false,"isExperimentalCompile":false,"dynamicIds":[84888],"gssp":true,"scriptLoader":[]}
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.