Advancing Dialectal Arabic to Modern Standard Arabic Machine Translation | Research Square window.SnipcartSettings = { analytics: { enabled: false } }; (function() { var accessVector = localStorage.getItem('access_vector') || ''; window.dataLayer = window.dataLayer || []; if (accessVector) { window.dataLayer.push({ user: { profile: { profileInfo: { snid: accessVector } } } }); } })(); (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src='https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-K279D39R'); Browse Preprints In Review Journals COVID-19 Preprints AJE Video Bytes Research Tools Research Promotion AJE Professional Editing AJE Rubriq About Preprint Platform In Review Editorial Policies Our Team Advisory Board Help Center Sign In Submit a Preprint Cite Share Download PDF Research Article Advancing Dialectal Arabic to Modern Standard Arabic Machine Translation Abdullah Alabdullah, Lifeng HAN, Chenghua Lin This is a preprint; it has not been peer reviewed by a journal. https://doi.org/ 10.21203/rs.3.rs-7510599/v1 This work is licensed under a CC BY 4.0 License Status: Under Revision Version 1 posted 13 You are reading this latest preprint version Abstract Dialectal Arabic (DA) poses a persistent challenge for natural language processing (NLP), as most everyday communication in the Arab world occurs in dialects that diverge significantly from Modern Standard Arabic (MSA). This linguistic divide impedes progress in Arabic machine translation. This paper presents two core contributions to advancing DA–MSA translation for the Levantine, Egyptian, and Gulf dialects, particularly in \textit{low-resource} and \textit{computationally constrained} settings: (i) a comprehensive evaluation of training-free prompting techniques, and (ii) the development of a resource-efficient fine-tuning pipeline. Our evaluation of \textbf{prompting} strategies across \textit{six} large language models (LLMs) found that few-shot prompting consistently outperformed zero-shot, chain-of-thought, and our proposed Ara-TEaR method. Ara-TEaR is designed as a three-stage self-refinement prompting process, targeting frequent meaning-transfer and adaptation errors in DA–MSA translation. In this evaluation, GPT-4o achieved the highest performance across all prompting settings. For \textbf{fine-tuning} LLMs, a quantized Gemma2-9B model achieved a chrF++ score of 49.88, outperforming zero-shot GPT-4o (44.58). Joint multi-dialect trained models outperformed single-dialect counterparts by over 10% chrF++, and 4-bit quantization reduced memory usage by 60% with less than 1% performance loss. The results and insights of our experiments offer a practical blueprint for improving dialectal inclusion in Arabic NLP, showing that high-quality DA–MSA machine translation is achievable even with limited resources and paving the way for more inclusive language technologies. Machine Translation Dialectal Arabic Modern Standard Arabic Translation Evaluation Large Language Model Fine-Tuning Full Text Additional Declarations No competing interests reported. Cite Share Download PDF Status: Under Revision Version 1 posted Editorial decision: Revision requested 27 Jan, 2026 Reviews received at journal 19 Jan, 2026 Reviewers agreed at journal 01 Jan, 2026 Reviewers agreed at journal 01 Jan, 2026 Reviews received at journal 09 Oct, 2025 Reviewers agreed at journal 09 Oct, 2025 Reviews received at journal 02 Oct, 2025 Reviewers agreed at journal 16 Sep, 2025 Reviewers agreed at journal 16 Sep, 2025 Reviewers invited by journal 16 Sep, 2025 Editor assigned by journal 16 Sep, 2025 Submission checks completed at journal 04 Sep, 2025 First submitted to journal 01 Sep, 2025 You are reading this latest preprint version Research Square lets you share your work early, gain feedback from the community, and start making changes to your manuscript prior to peer review in a journal. As a division of Research Square Company, we’re committed to making research communication faster, fairer, and more useful. We do this by developing innovative software and high quality services for the global research community. Our growing team is made up of researchers and industry professionals working together to solve the most critical problems facing scientific publishing. Also discoverable on Platform About Our Team In Review Editorial Policies Advisory Board Help Center Resources Author Services Accessibility API Access RSS feed Manage Cookie Preferences © Research Square 2026 | ISSN 2693-5015 (online) Privacy Policy Terms of Service Do Not Sell My Personal Information {"props":{"pageProps":{"initialData":{"identity":"rs-7510599","acceptedTermsAndConditions":true,"allowDirectSubmit":false,"archivedVersions":[],"articleType":"Research Article","associatedPublications":[],"authors":[{"id":576962596,"identity":"d3a2da2a-ae61-4ec7-9511-e024de252238","order_by":0,"name":"Abdullah Alabdullah","email":"","orcid":"","institution":"University of Manchester","correspondingAuthor":false,"prefix":"","firstName":"Abdullah","middleName":"","lastName":"Alabdullah","suffix":""},{"id":576962597,"identity":"70402cb7-d982-414d-81b3-7edfbc6c19e2","order_by":1,"name":"Lifeng HAN","email":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAZAAAAAyAQMAAABI0h/eAAAABlBMVEX///8AAABVwtN+AAAACXBIWXMAAA7EAAAOxAGVKw4bAAAAk0lEQVRIiWNgGAWjYBACPgYeIFlhAeE9IEYLG1jLGQkIL4FoLYxtpGnhPfi5cJ5E4tr2A2wPiNTClyw9c5tE4rYzCewGxDrMQJoXpOUGA5sEsVqMf/POIVGLmTRvA0lamHnMrHmOSRhvO5PYRpwWfvYe49s8NTay244fPibxgRgtDMxwFmMDURpGwSgYBaNgFBABANTVJp0++eVmAAAAAElFTkSuQmCC","orcid":"","institution":"Leiden University Medical Center","correspondingAuthor":true,"prefix":"","firstName":"Lifeng","middleName":"","lastName":"HAN","suffix":""},{"id":576962598,"identity":"7c743e8e-a5af-4c1c-8b03-3727fd99e913","order_by":2,"name":"Chenghua Lin","email":"","orcid":"","institution":"University of Manchester","correspondingAuthor":false,"prefix":"","firstName":"Chenghua","middleName":"","lastName":"Lin","suffix":""}],"badges":[],"createdAt":"2025-09-01 17:08:14","currentVersionCode":1,"declarations":"","doi":"10.21203/rs.3.rs-7510599/v1","doiUrl":"https://doi.org/10.21203/rs.3.rs-7510599/v1","draftVersion":[],"editorialEvents":[],"editorialNote":"","failedWorkflow":false,"files":[{"id":100951276,"identity":"a88bedaf-bef4-4366-b2f1-22144a9b8d6c","added_by":"auto","created_at":"2026-01-23 07:10:22","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":1100321,"visible":true,"origin":"","legend":"","description":"","filename":"JLREofficialtemplate2025Copy4ArabMTCopy4submission.pdf","url":"https://assets-eu.researchsquare.com/files/rs-7510599/v1_covered_90221df6-9098-478f-9810-caa3d237543b.pdf"}],"financialInterests":"No competing interests reported.","formattedTitle":"Advancing Dialectal Arabic to Modern Standard Arabic Machine Translation","fulltext":[],"fulltextSource":"","fullText":"","funders":[],"hasAdminPriorityOnWorkflow":false,"hasManuscriptDocX":false,"hasOptedInToPreprint":true,"hasPassedJournalQc":"","hasAnyPriority":false,"hideJournal":false,"highlight":"","institution":"","isAcceptedByJournal":false,"isAuthorSuppliedPdf":true,"isDeskRejected":"","isHiddenFromSearch":false,"isInQc":false,"isInWorkflow":false,"isPdf":true,"isPdfUpToDate":true,"isWithdrawnOrRetracted":false,"journal":{"display":true,"email":"
[email protected]","identity":"language-resources-and-evaluation","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":false,"externalIdentity":"lrev","sideBox":"Learn more about [Language Resources and Evaluation](http://link.springer.com/journal/10579)","snPcode":"10579","submissionUrl":"https://submission.nature.com/new-submission/10579/3","title":"Language Resources and Evaluation","twitterHandle":"","acdcEnabled":true,"dfaEnabled":true,"editorialSystem":"em","reportingPortfolio":"Springer Hybrid","inReviewEnabled":true,"inReviewRevisionsEnabled":false},"keywords":"Machine Translation, Dialectal Arabic, Modern Standard Arabic, Translation Evaluation, Large Language Model, Fine-Tuning","lastPublishedDoi":"10.21203/rs.3.rs-7510599/v1","lastPublishedDoiUrl":"https://doi.org/10.21203/rs.3.rs-7510599/v1","license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"manuscriptAbstract":"Dialectal Arabic (DA) poses a persistent challenge for natural language processing (NLP), as most everyday communication in the Arab world occurs in dialects that diverge significantly from Modern Standard Arabic (MSA). This linguistic divide impedes progress in Arabic machine translation.\nThis paper presents two core contributions to advancing DA–MSA translation for the Levantine, Egyptian, and Gulf dialects, particularly in \\textit{low-resource} and \\textit{computationally constrained} settings: (i) a comprehensive evaluation of training-free prompting techniques, and (ii) the development of a resource-efficient fine-tuning pipeline.\nOur evaluation of \\textbf{prompting} strategies across \\textit{six} large language models (LLMs) found that few-shot prompting consistently outperformed zero-shot, chain-of-thought, and our proposed Ara-TEaR method. Ara-TEaR is designed as a three-stage self-refinement prompting process, targeting frequent meaning-transfer and adaptation errors in DA–MSA translation. In this evaluation, GPT-4o achieved the highest performance across all prompting settings.\nFor \\textbf{fine-tuning} LLMs, a quantized Gemma2-9B model achieved a chrF++ score of 49.88, outperforming zero-shot GPT-4o (44.58). Joint multi-dialect trained models outperformed single-dialect counterparts by over 10\\% chrF++, and 4-bit quantization reduced memory usage by 60\\% with less than 1\\% performance loss.\nThe results and insights of our experiments offer a practical blueprint for improving dialectal inclusion in Arabic NLP, showing that high-quality DA–MSA machine translation is achievable even with limited resources and paving the way for more inclusive language technologies.","manuscriptTitle":"Advancing Dialectal Arabic to Modern Standard Arabic Machine Translation","msid":"","msnumber":"","nonDraftVersions":[{"code":1,"date":"2026-01-22 15:22:56","doi":"10.21203/rs.3.rs-7510599/v1","editorialEvents":[{"type":"communityComments","content":0},{"type":"decision","content":"Revision requested","date":"2026-01-27T15:41:24+00:00","index":"","fulltext":""},{"type":"editorInvitedReview","content":"","date":"2026-01-19T10:37:14+00:00","index":"hide","fulltext":""},{"type":"reviewerAgreed","content":"330333007369302305052474018734952220998","date":"2026-01-02T04:12:29+00:00","index":"hide","fulltext":""},{"type":"reviewerAgreed","content":"104015470693282124707906542311956826958","date":"2026-01-01T13:23:02+00:00","index":"hide","fulltext":""},{"type":"editorInvitedReview","content":"","date":"2025-10-09T21:22:01+00:00","index":"hide","fulltext":""},{"type":"reviewerAgreed","content":"38074950565926025315044840676105541255","date":"2025-10-09T16:44:09+00:00","index":"hide","fulltext":""},{"type":"editorInvitedReview","content":"","date":"2025-10-02T22:16:58+00:00","index":"hide","fulltext":""},{"type":"reviewerAgreed","content":"39696844855720644068835032951690844088","date":"2025-09-16T09:33:01+00:00","index":"hide","fulltext":""},{"type":"reviewerAgreed","content":"44827057251698475498148835369374725933","date":"2025-09-16T09:12:36+00:00","index":"hide","fulltext":""},{"type":"reviewersInvited","content":"","date":"2025-09-16T09:01:46+00:00","index":"","fulltext":""},{"type":"editorAssigned","content":"","date":"2025-09-16T08:50:50+00:00","index":"","fulltext":""},{"type":"checksComplete","content":"","date":"2025-09-05T02:03:02+00:00","index":"","fulltext":""},{"type":"submitted","content":"Language Resources and Evaluation","date":"2025-09-01T16:56:14+00:00","index":"","fulltext":""}],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"language-resources-and-evaluation","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":false,"externalIdentity":"lrev","sideBox":"Learn more about [Language Resources and Evaluation](http://link.springer.com/journal/10579)","snPcode":"10579","submissionUrl":"https://submission.nature.com/new-submission/10579/3","title":"Language Resources and Evaluation","twitterHandle":"","acdcEnabled":true,"dfaEnabled":true,"editorialSystem":"em","reportingPortfolio":"Springer Hybrid","inReviewEnabled":true,"inReviewRevisionsEnabled":false}}],"origin":"","ownerIdentity":"7302833b-6d79-4b24-b259-4e3f2ee1c982","owner":[],"postedDate":"January 22nd, 2026","published":true,"recentEditorialEvents":[],"rejectedJournal":[],"revision":"","amendment":"","status":"in-revision","subjectAreas":[],"tags":[],"updatedAt":"2026-01-27T15:56:59+00:00","versionOfRecord":[],"versionCreatedAt":"2026-01-22 15:22:56","video":"","vorDoi":"","vorDoiUrl":"","workflowStages":[]},"version":"v1","identity":"rs-7510599","journalConfig":"researchsquare"},"__N_SSP":true},"page":"/article/[identity]/[[...version]]","query":{"redirect":"/article/rs-7510599","identity":"rs-7510599","version":["v1"]},"buildId":"XKTyCvWXoU3ODBz1xrDgd","isFallback":false,"isExperimentalCompile":false,"dynamicIds":[84888],"gssp":true,"scriptLoader":[]}
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.