Translating dialects between ChatGPT and... | F1000Research "use strict";function _typeof(t){return(_typeof="function"==typeof Symbol&&"symbol"==typeof Symbol.iterator?function(t){return typeof t}:function(t){return t&&"function"==typeof Symbol&&t.constructor===Symbol&&t!==Symbol.prototype?"symbol":typeof t})(t)}!function(){var t=function(){var t,e,o=[],n=window,r=n;for(;r;){try{if(r.frames.__tcfapiLocator){t=r;break}}catch(t){}if(r===n.top)break;r=r.parent}t||(!function t(){var e=n.document,o=!!n.frames.__tcfapiLocator;if(!o)if(e.body){var r=e.createElement("iframe");r.style.cssText="display:none",r.name="__tcfapiLocator",e.body.appendChild(r)}else setTimeout(t,5);return!o}(),n.__tcfapi=function(){for(var t=arguments.length,n=new Array(t),r=0;r 3&&2===parseInt(n[1],10)&&"boolean"==typeof n[3]&&(e=n[3],"function"==typeof n[2]&&n[2]("set",!0)):"ping"===n[0]?"function"==typeof n[2]&&n[2]({gdprApplies:e,cmpLoaded:!1,cmpStatus:"stub"}):o.push(n)},n.addEventListener("message",(function(t){var e="string"==typeof t.data,o={};if(e)try{o=JSON.parse(t.data)}catch(t){}else o=t.data;var n="object"===_typeof(o)&&null!==o?o.__tcfapiCall:null;n&&window.__tcfapi(n.command,n.version,(function(o,r){var a={__tcfapiReturn:{returnValue:o,success:r,callId:n.callId}};t&&t.source&&t.source.postMessage&&t.source.postMessage(e?JSON.stringify(a):a,"*")}),n.parameter)}),!1))};"undefined"!=typeof module?module.exports=t:t()}(); dataLayer = dataLayer || []; // Standard GTM initialization - Google Consent Mode handles consent automatically (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start': new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0], j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src= 'https://www.googletagmanager.com/gtm.js?id='+i+dl+ '>m_auth=hzk0Vc3qFsQYhCrIoHz68A>m_preview=env-1>m_cookies_win=x';f.parentNode.insertBefore(j,f); })(window,document,'script','dataLayer','GTM-MWFK8L5J'); ;window.NREUM||(NREUM={});NREUM.init={distributed_tracing:{enabled:true},privacy:{cookies_enabled:true},ajax:{deny_list:["bam.nr-data.net"]}}; ;NREUM.loader_config={accountID:"438030",trustKey:"438030",agentID:"772317073",licenseKey:"97f8f67f26",applicationID:"772317073"} ;NREUM.info={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net",licenseKey:"97f8f67f26",applicationID:"772317073",sa:1} ;/*! For license information please see nr-loader-spa-1.236.0.min.js.LICENSE.txt */ (()=>{"use strict";var e,t,r={5763:(e,t,r)=>{r.d(t,{P_:()=>l,Mt:()=>g,C5:()=>s,DL:()=>v,OP:()=>T,lF:()=>D,Yu:()=>y,Dg:()=>h,CX:()=>c,GE:()=>b,sU:()=>_});var n=r(8632),i=r(9567);const o={beacon:n.ce.beacon,errorBeacon:n.ce.errorBeacon,licenseKey:void 0,applicationID:void 0,sa:void 0,queueTime:void 0,applicationTime:void 0,ttGuid:void 0,user:void 0,account:void 0,product:void 0,extra:void 0,jsAttributes:{},userAttributes:void 0,atts:void 0,transactionName:void 0,tNamePlain:void 0},a={};function s(e){if(!e)throw new Error("All info objects require an agent identifier!");if(!a[e])throw new Error("Info for ".concat(e," was never set"));return a[e]}function c(e,t){if(!e)throw new Error("All info objects require an agent identifier!");a[e]=(0,i.D)(t,o),(0,n.Qy)(e,a[e],"info")}var u=r(7056);const d=()=>{const e={blockSelector:"[data-nr-block]",maskInputOptions:{password:!0}};return{allow_bfcache:!0,privacy:{cookies_enabled:!0},ajax:{deny_list:void 0,enabled:!0,harvestTimeSeconds:10},distributed_tracing:{enabled:void 0,exclude_newrelic_header:void 0,cors_use_newrelic_header:void 0,cors_use_tracecontext_headers:void 0,allowed_origins:void 0},session:{domain:void 0,expiresMs:u.oD,inactiveMs:u.Hb},ssl:void 0,obfuscate:void 0,jserrors:{enabled:!0,harvestTimeSeconds:10},metrics:{enabled:!0},page_action:{enabled:!0,harvestTimeSeconds:30},page_view_event:{enabled:!0},page_view_timing:{enabled:!0,harvestTimeSeconds:30,long_task:!1},session_trace:{enabled:!0,harvestTimeSeconds:10},harvest:{tooManyRequestsDelay:60},session_replay:{enabled:!1,harvestTimeSeconds:60,sampleRate:.1,errorSampleRate:.1,maskTextSelector:"*",maskAllInputs:!0,get blockClass(){return"nr-block"},get ignoreClass(){return"nr-ignore"},get maskTextClass(){return"nr-mask"},get blockSelector(){return e.blockSelector},set blockSelector(t){e.blockSelector+=",".concat(t)},get maskInputOptions(){return e.maskInputOptions},set maskInputOptions(t){e.maskInputOptions={...t,password:!0}}},spa:{enabled:!0,harvestTimeSeconds:10}}},f={};function l(e){if(!e)throw new Error("All configuration objects require an agent identifier!");if(!f[e])throw new Error("Configuration for ".concat(e," was never set"));return f[e]}function h(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");f[e]=(0,i.D)(t,d()),(0,n.Qy)(e,f[e],"config")}function g(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");var r=l(e);if(r){for(var n=t.split("."),i=0;i {r.d(t,{D:()=>i});var n=r(50);function i(e,t){try{if(!e||"object"!=typeof e)return(0,n.Z)("Setting a Configurable requires an object as input");if(!t||"object"!=typeof t)return(0,n.Z)("Setting a Configurable requires a model to set its initial properties");const r=Object.create(Object.getPrototypeOf(t),Object.getOwnPropertyDescriptors(t)),o=0===Object.keys(r).length?e:r;for(let a in o)if(void 0!==e[a])try{"object"==typeof e[a]&&"object"==typeof t[a]?r[a]=i(e[a],t[a]):r[a]=e[a]}catch(e){(0,n.Z)("An error occurred while setting a property of a Configurable",e)}return r}catch(e){(0,n.Z)("An error occured while setting a Configurable",e)}}},6818:(e,t,r)=>{r.d(t,{Re:()=>i,gF:()=>o,q4:()=>n});const n="1.236.0",i="PROD",o="CDN"},385:(e,t,r)=>{r.d(t,{FN:()=>a,IF:()=>u,Nk:()=>f,Tt:()=>s,_A:()=>o,il:()=>n,pL:()=>c,v6:()=>i,w1:()=>d});const n="undefined"!=typeof window&&!!window.document,i="undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self.navigator instanceof WorkerNavigator||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis.navigator instanceof WorkerNavigator),o=n?window:"undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis),a=""+o?.location,s=/iPad|iPhone|iPod/.test(navigator.userAgent),c=s&&"undefined"==typeof SharedWorker,u=(()=>{const e=navigator.userAgent.match(/Firefox[/\s](\d+\.\d+)/);return Array.isArray(e)&&e.length>=2?+e[1]:0})(),d=Boolean(n&&window.document.documentMode),f=!!navigator.sendBeacon},1117:(e,t,r)=>{r.d(t,{w:()=>o});var n=r(50);const i={agentIdentifier:"",ee:void 0};class o{constructor(e){try{if("object"!=typeof e)return(0,n.Z)("shared context requires an object as input");this.sharedContext={},Object.assign(this.sharedContext,i),Object.entries(e).forEach((e=>{let[t,r]=e;Object.keys(i).includes(t)&&(this.sharedContext[t]=r)}))}catch(e){(0,n.Z)("An error occured while setting SharedContext",e)}}}},8e3:(e,t,r)=>{r.d(t,{L:()=>d,R:()=>c});var n=r(2177),i=r(1284),o=r(4322),a=r(3325);const s={};function c(e,t){const r={staged:!1,priority:a.p[t]||0};u(e),s[e].get(t)||s[e].set(t,r)}function u(e){e&&(s[e]||(s[e]=new Map))}function d(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"",t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:"feature";if(u(e),!e||!s[e].get(t))return a(t);s[e].get(t).staged=!0;const r=[...s[e]];function a(t){const r=e?n.ee.get(e):n.ee,a=o.X.handlers;if(r.backlog&&a){var s=r.backlog[t],c=a[t];if(c){for(var u=0;s&&u {let[t,r]=e;return r.staged}))&&(r.sort(((e,t)=>e[1].priority-t[1].priority)),r.forEach((e=>{let[t]=e;a(t)})))}function f(e,t){var r=e[1];(0,i.D)(t[r],(function(t,r){var n=e[0];if(r[0]===n){var i=r[1],o=e[3],a=e[2];i.apply(o,a)}}))}},2177:(e,t,r)=>{r.d(t,{c:()=>f,ee:()=>u});var n=r(8632),i=r(2210),o=r(1284),a=r(5763),s="nr@context";let c=(0,n.fP)();var u;function d(){}function f(e){return(0,i.X)(e,s,l)}function l(){return new d}function h(){u.aborted=!0,u.backlog={}}c.ee?u=c.ee:(u=function e(t,r){var n={},c={},f={},g=!1;try{g=16===r.length&&(0,a.OP)(r).isolatedBacklog}catch(e){}var p={on:b,addEventListener:b,removeEventListener:y,emit:v,get:x,listeners:w,context:m,buffer:A,abort:h,aborted:!1,isBuffering:E,debugId:r,backlog:g?{}:t&&"object"==typeof t.backlog?t.backlog:{}};return p;function m(e){return e&&e instanceof d?e:e?(0,i.X)(e,s,l):l()}function v(e,r,n,i,o){if(!1!==o&&(o=!0),!u.aborted||i){t&&o&&t.emit(e,r,n);for(var a=m(n),s=w(e),d=s.length,f=0;fn,p:()=>i});var n=r(2177).ee.get("handle");function i(e,t,r,i,o){o?(o.buffer([e],i),o.emit(e,t,r)):(n.buffer([e],i),n.emit(e,t,r))}},4322:(e,t,r)=>{r.d(t,{X:()=>o});var n=r(5546);o.on=a;var i=o.handlers={};function o(e,t,r,o){a(o||n.E,i,e,t,r)}function a(e,t,r,i,o){o||(o="feature"),e||(e=n.E);var a=t[o]=t[o]||{};(a[r]=a[r]||[]).push([e,i])}},3239:(e,t,r)=>{r.d(t,{bP:()=>s,iz:()=>c,m$:()=>a});var n=r(385);let i=!1,o=!1;try{const e={get passive(){return i=!0,!1},get signal(){return o=!0,!1}};n._A.addEventListener("test",null,e),n._A.removeEventListener("test",null,e)}catch(e){}function a(e,t){return i||o?{capture:!!e,passive:i,signal:t}:!!e}function s(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;window.addEventListener(e,t,a(r,n))}function c(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;document.addEventListener(e,t,a(r,n))}},4402:(e,t,r)=>{r.d(t,{Ht:()=>u,M:()=>c,Rl:()=>a,ky:()=>s});var n=r(385);const i="xxxxxxxx-xxxx-4xxx-yxxx-xxxxxxxxxxxx";function o(e,t){return e?15&e[t]:16*Math.random()|0}function a(){const e=n._A?.crypto||n._A?.msCrypto;let t,r=0;return e&&e.getRandomValues&&(t=e.getRandomValues(new Uint8Array(31))),i.split("").map((e=>"x"===e?o(t,++r).toString(16):"y"===e?(3&o()|8).toString(16):e)).join("")}function s(e){const t=n._A?.crypto||n._A?.msCrypto;let r,i=0;t&&t.getRandomValues&&(r=t.getRandomValues(new Uint8Array(31)));const a=[];for(var s=0;s {r.d(t,{Bq:()=>n,Hb:()=>o,oD:()=>i});const n="NRBA",i=144e5,o=18e5},7894:(e,t,r)=>{function n(){return Math.round(performance.now())}r.d(t,{z:()=>n})},7243:(e,t,r)=>{r.d(t,{e:()=>o});var n=r(385),i={};function o(e){if(e in i)return i[e];if(0===(e||"").indexOf("data:"))return{protocol:"data"};let t;var r=n._A?.location,o={};if(n.il)t=document.createElement("a"),t.href=e;else try{t=new URL(e,r.href)}catch(e){return o}o.port=t.port;var a=t.href.split("://");!o.port&&a[1]&&(o.port=a[1].split("/")[0].split("@").pop().split(":")[1]),o.port&&"0"!==o.port||(o.port="https"===a[0]?"443":"80"),o.hostname=t.hostname||r.hostname,o.pathname=t.pathname,o.protocol=a[0],"/"!==o.pathname.charAt(0)&&(o.pathname="/"+o.pathname);var s=!t.protocol||":"===t.protocol||t.protocol===r.protocol,c=t.hostname===r.hostname&&t.port===r.port;return o.sameOrigin=s&&(!t.hostname||c),"/"===o.pathname&&(i[e]=o),o}},50:(e,t,r)=>{function n(e,t){"function"==typeof console.warn&&(console.warn("New Relic: ".concat(e)),t&&console.warn(t))}r.d(t,{Z:()=>n})},2587:(e,t,r)=>{r.d(t,{N:()=>c,T:()=>u});var n=r(2177),i=r(5546),o=r(8e3),a=r(3325);const s={stn:[a.D.sessionTrace],err:[a.D.jserrors,a.D.metrics],ins:[a.D.pageAction],spa:[a.D.spa],sr:[a.D.sessionReplay,a.D.sessionTrace]};function c(e,t){const r=n.ee.get(t);e&&"object"==typeof e&&(Object.entries(e).forEach((e=>{let[t,n]=e;void 0===u[t]&&(s[t]?s[t].forEach((e=>{n?(0,i.p)("feat-"+t,[],void 0,e,r):(0,i.p)("block-"+t,[],void 0,e,r),(0,i.p)("rumresp-"+t,[Boolean(n)],void 0,e,r)})):n&&(0,i.p)("feat-"+t,[],void 0,void 0,r),u[t]=Boolean(n))})),Object.keys(s).forEach((e=>{void 0===u[e]&&(s[e]?.forEach((t=>(0,i.p)("rumresp-"+e,[!1],void 0,t,r))),u[e]=!1)})),(0,o.L)(t,a.D.pageViewEvent))}const u={}},2210:(e,t,r)=>{r.d(t,{X:()=>i});var n=Object.prototype.hasOwnProperty;function i(e,t,r){if(n.call(e,t))return e[t];var i=r();if(Object.defineProperty&&Object.keys)try{return Object.defineProperty(e,t,{value:i,writable:!0,enumerable:!1}),i}catch(e){}return e[t]=i,i}},1284:(e,t,r)=>{r.d(t,{D:()=>n});const n=(e,t)=>Object.entries(e||{}).map((e=>{let[r,n]=e;return t(r,n)}))},4351:(e,t,r)=>{r.d(t,{P:()=>o});var n=r(2177);const i=()=>{const e=new WeakSet;return(t,r)=>{if("object"==typeof r&&null!==r){if(e.has(r))return;e.add(r)}return r}};function o(e){try{return JSON.stringify(e,i())}catch(e){try{n.ee.emit("internal-error",[e])}catch(e){}}}},3960:(e,t,r)=>{r.d(t,{K:()=>a,b:()=>o});var n=r(3239);function i(){return"undefined"==typeof document||"complete"===document.readyState}function o(e,t){if(i())return e();(0,n.bP)("load",e,t)}function a(e){if(i())return e();(0,n.iz)("DOMContentLoaded",e)}},8632:(e,t,r)=>{r.d(t,{EZ:()=>u,Qy:()=>c,ce:()=>o,fP:()=>a,gG:()=>d,mF:()=>s});var n=r(7894),i=r(385);const o={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net"};function a(){return i._A.NREUM||(i._A.NREUM={}),void 0===i._A.newrelic&&(i._A.newrelic=i._A.NREUM),i._A.NREUM}function s(){let e=a();return e.o||(e.o={ST:i._A.setTimeout,SI:i._A.setImmediate,CT:i._A.clearTimeout,XHR:i._A.XMLHttpRequest,REQ:i._A.Request,EV:i._A.Event,PR:i._A.Promise,MO:i._A.MutationObserver,FETCH:i._A.fetch}),e}function c(e,t,r){let i=a();const o=i.initializedAgents||{},s=o[e]||{};return Object.keys(s).length||(s.initializedAt={ms:(0,n.z)(),date:new Date}),i.initializedAgents={...o,[e]:{...s,[r]:t}},i}function u(e,t){a()[e]=t}function d(){return function(){let e=a();const t=e.info||{};e.info={beacon:o.beacon,errorBeacon:o.errorBeacon,...t}}(),function(){let e=a();const t=e.init||{};e.init={...t}}(),s(),function(){let e=a();const t=e.loader_config||{};e.loader_config={...t}}(),a()}},7956:(e,t,r)=>{r.d(t,{N:()=>i});var n=r(3239);function i(e){let t=arguments.length>1&&void 0!==arguments[1]&&arguments[1],r=arguments.length>2?arguments[2]:void 0,i=arguments.length>3?arguments[3]:void 0;return void(0,n.iz)("visibilitychange",(function(){if(t)return void("hidden"==document.visibilityState&&e());e(document.visibilityState)}),r,i)}},1214:(e,t,r)=>{r.d(t,{em:()=>v,u5:()=>N,QU:()=>S,_L:()=>I,Gm:()=>L,Lg:()=>M,gy:()=>U,BV:()=>Q,Kf:()=>ee});var n=r(2177);const i="nr@original";var o=Object.prototype.hasOwnProperty,a=!1;function s(e,t){return e||(e=n.ee),r.inPlace=function(e,t,n,i,o){n||(n="");var a,s,c,u="-"===n.charAt(0);for(c=0;c 2?n-2:0),o=2;o {r(A[T],e,w),r(E[T],e,w)})),r(l._A,"fetch",y),t.on(y+"end",(function(e,r){var n=this;if(r){var i=r.headers.get("content-length");null!==i&&(n.rxSize=i),t.emit(y+"done",[null,r],n)}else t.emit(y+"done",[e],n)})),t}const O={},j=["pushState","replaceState"];function S(e){const t=function(e){return(e||n.ee).get("history")}(e);return!l.il||O[t.debugId]++||(O[t.debugId]=1,s(t).inPlace(window.history,j,"-")),t}var P=r(3239);const C={},R=["appendChild","insertBefore","replaceChild"];function I(e){const t=function(e){return(e||n.ee).get("jsonp")}(e);if(!l.il||C[t.debugId])return t;C[t.debugId]=!0;var r=s(t),i=/[?&](?:callback|cb)=([^&#]+)/,o=/(.*)\.([^.]+)/,a=/^(\w+)(\.|$)(.*)$/;function c(e,t){var r=e.match(a),n=r[1],i=r[3];return i?c(i,t[n]):t[n]}return r.inPlace(Node.prototype,R,"dom-"),t.on("dom-start",(function(e){!function(e){if(!e||"string"!=typeof e.nodeName||"script"!==e.nodeName.toLowerCase())return;if("function"!=typeof e.addEventListener)return;var n=(a=e.src,s=a.match(i),s?s[1]:null);var a,s;if(!n)return;var u=function(e){var t=e.match(o);if(t&&t.length>=3)return{key:t[2],parent:c(t[1],window)};return{key:e,parent:window}}(n);if("function"!=typeof u.parent[u.key])return;var d={};function f(){t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}function l(){t.emit("jsonp-error",[],d),t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}r.inPlace(u.parent,[u.key],"cb-",d),e.addEventListener("load",f,(0,P.m$)(!1)),e.addEventListener("error",l,(0,P.m$)(!1)),t.emit("new-jsonp",[e.src],d)}(e[0])})),t}var k=r(5763);const H={};function L(e){const t=function(e){return(e||n.ee).get("mutation")}(e);if(!l.il||H[t.debugId])return t;H[t.debugId]=!0;var r=s(t),i=k.Yu.MO;return i&&(window.MutationObserver=function(e){return this instanceof i?new i(r(e,"fn-")):i.apply(this,arguments)},MutationObserver.prototype=i.prototype),t}const z={};function M(e){const t=function(e){return(e||n.ee).get("promise")}(e);if(z[t.debugId])return t;z[t.debugId]=!0;var r=n.c,o=s(t),a=k.Yu.PR;return a&&function(){function e(r){var n=t.context(),i=o(r,"executor-",n,null,!1);const s=Reflect.construct(a,[i],e);return t.context(s).getCtx=function(){return n},s}l._A.Promise=e,Object.defineProperty(e,"name",{value:"Promise"}),e.toString=function(){return a.toString()},Object.setPrototypeOf(e,a),["all","race"].forEach((function(r){const n=a[r];e[r]=function(e){let i=!1;[...e||[]].forEach((e=>{this.resolve(e).then(a("all"===r),a(!1))}));const o=n.apply(this,arguments);return o;function a(e){return function(){t.emit("propagate",[null,!i],o,!1,!1),i=i||!e}}}})),["resolve","reject"].forEach((function(r){const n=a[r];e[r]=function(e){const r=n.apply(this,arguments);return e!==r&&t.emit("propagate",[e,!0],r,!1,!1),r}})),e.prototype=a.prototype;const n=a.prototype.then;a.prototype.then=function(){var e=this,i=r(e);i.promise=e;for(var a=arguments.length,s=new Array(a),c=0;c e())),t};function m(e,t){i.inPlace(t,["onreadystatechange"],"fn-",E)}function b(){var e=this,t=r.context(e);e.readyState>3&&!t.resolved&&(t.resolved=!0,r.emit("xhr-resolved",[],e)),i.inPlace(e,f,"fn-",E)}if(function(e,t){for(var r in e)t[r]=e[r]}(o,p),p.prototype=o.prototype,i.inPlace(p.prototype,J,"-xhr-",E),r.on("send-xhr-start",(function(e,t){m(e,t),function(e){h.push(e),a&&(y?y.then(A):u?u(A):(w=-w,x.data=w))}(t)})),r.on("open-xhr-start",m),a){var y=c&&c.resolve();if(!u&&!c){var w=1,x=document.createTextNode(w);new a(A).observe(x,{characterData:!0})}}else t.on("fn-end",(function(e){e[0]&&e[0].type===d||A()}));function A(){for(var e=0;e {r.d(t,{t:()=>n});const n=r(3325).D.ajax},6660:(e,t,r)=>{r.d(t,{A:()=>i,t:()=>n});const n=r(3325).D.jserrors,i="nr@seenError"},3081:(e,t,r)=>{r.d(t,{gF:()=>o,mY:()=>i,t9:()=>n,vz:()=>s,xS:()=>a});const n=r(3325).D.metrics,i="sm",o="cm",a="storeSupportabilityMetrics",s="storeEventMetrics"},4649:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageAction},7633:(e,t,r)=>{r.d(t,{Dz:()=>i,OJ:()=>a,qw:()=>o,t9:()=>n});const n=r(3325).D.pageViewEvent,i="firstbyte",o="domcontent",a="windowload"},9251:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageViewTiming},3614:(e,t,r)=>{r.d(t,{BST_RESOURCE:()=>i,END:()=>s,FEATURE_NAME:()=>n,FN_END:()=>u,FN_START:()=>c,PUSH_STATE:()=>d,RESOURCE:()=>o,START:()=>a});const n=r(3325).D.sessionTrace,i="bstResource",o="resource",a="-start",s="-end",c="fn"+a,u="fn"+s,d="pushState"},7836:(e,t,r)=>{r.d(t,{BODY:()=>A,CB_END:()=>E,CB_START:()=>u,END:()=>x,FEATURE_NAME:()=>i,FETCH:()=>_,FETCH_BODY:()=>v,FETCH_DONE:()=>m,FETCH_START:()=>p,FN_END:()=>c,FN_START:()=>s,INTERACTION:()=>l,INTERACTION_API:()=>d,INTERACTION_EVENTS:()=>o,JSONP_END:()=>b,JSONP_NODE:()=>g,JS_TIME:()=>T,MAX_TIMER_BUDGET:()=>a,REMAINING:()=>f,SPA_NODE:()=>h,START:()=>w,originalSetTimeout:()=>y});var n=r(5763);const i=r(3325).D.spa,o=["click","submit","keypress","keydown","keyup","change"],a=999,s="fn-start",c="fn-end",u="cb-start",d="api-ixn-",f="remaining",l="interaction",h="spaNode",g="jsonpNode",p="fetch-start",m="fetch-done",v="fetch-body-",b="jsonp-end",y=n.Yu.ST,w="-start",x="-end",A="-body",E="cb"+x,T="jsTime",_="fetch"},5938:(e,t,r)=>{r.d(t,{W:()=>o});var n=r(5763),i=r(2177);class o{constructor(e,t,r){this.agentIdentifier=e,this.aggregator=t,this.ee=i.ee.get(e,(0,n.OP)(this.agentIdentifier).isolatedBacklog),this.featureName=r,this.blocked=!1}}},9144:(e,t,r)=>{r.d(t,{j:()=>m});var n=r(3325),i=r(5763),o=r(5546),a=r(2177),s=r(7894),c=r(8e3),u=r(3960),d=r(385),f=r(50),l=r(3081),h=r(8632);function g(){const e=(0,h.gG)();["setErrorHandler","finished","addToTrace","inlineHit","addRelease","addPageAction","setCurrentRouteName","setPageViewName","setCustomAttribute","interaction","noticeError","setUserId"].forEach((t=>{e[t]=function(){for(var r=arguments.length,n=new Array(r),i=0;i 1?r-1:0),i=1;i {e.exposed&&e.api[t]&&o.push(e.api[t](...n))})),o.length>1?o:o[0]}(t,...n)}}))}var p=r(2587);function m(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:{},m=arguments.length>2?arguments[2]:void 0,v=arguments.length>3?arguments[3]:void 0,{init:b,info:y,loader_config:w,runtime:x={loaderType:m},exposed:A=!0}=t;const E=(0,h.gG)();y||(b=E.init,y=E.info,w=E.loader_config),(0,i.Dg)(e,b||{}),(0,i.GE)(e,w||{}),(0,i.sU)(e,x),y.jsAttributes??={},d.v6&&(y.jsAttributes.isWorker=!0),(0,i.CX)(e,y),g();const T=function(e,t){t||(0,c.R)(e,"api");const h={};var g=a.ee.get(e),p=g.get("tracer"),m="api-",v=m+"ixn-";function b(t,r,n,o){const a=(0,i.C5)(e);return null===r?delete a.jsAttributes[t]:(0,i.CX)(e,{...a,jsAttributes:{...a.jsAttributes,[t]:r}}),x(m,n,!0,o||null===r?"session":void 0)(t,r)}function y(){}["setErrorHandler","finished","addToTrace","inlineHit","addRelease"].forEach((e=>h[e]=x(m,e,!0,"api"))),h.addPageAction=x(m,"addPageAction",!0,n.D.pageAction),h.setCurrentRouteName=x(m,"routeName",!0,n.D.spa),h.setPageViewName=function(t,r){if("string"==typeof t)return"/"!==t.charAt(0)&&(t="/"+t),(0,i.OP)(e).customTransaction=(r||"http://custom.transaction")+t,x(m,"setPageViewName",!0)()},h.setCustomAttribute=function(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2];if("string"==typeof e){if(["string","number"].includes(typeof t)||null===t)return b(e,t,"setCustomAttribute",r);(0,f.Z)("Failed to execute setCustomAttribute.\nNon-null value must be a string or number type, but a type of was provided."))}else(0,f.Z)("Failed to execute setCustomAttribute.\nName must be a string type, but a type of was provided."))},h.setUserId=function(e){if("string"==typeof e||null===e)return b("enduser.id",e,"setUserId",!0);(0,f.Z)("Failed to execute setUserId.\nNon-null value must be a string type, but a type of was provided."))},h.interaction=function(){return(new y).get()};var w=y.prototype={createTracer:function(e,t){var r={},i=this,a="function"==typeof t;return(0,o.p)(v+"tracer",[(0,s.z)(),e,r],i,n.D.spa,g),function(){if(p.emit((a?"":"no-")+"fn-start",[(0,s.z)(),i,a],r),a)try{return t.apply(this,arguments)}catch(e){throw p.emit("fn-err",[arguments,this,"string"==typeof e?new Error(e):e],r),e}finally{p.emit("fn-end",[(0,s.z)()],r)}}}};function x(e,t,r,i){return function(){return(0,o.p)(l.xS,["API/"+t+"/called"],void 0,n.D.metrics,g),i&&(0,o.p)(e+t,[(0,s.z)(),...arguments],r?null:this,i,g),r?void 0:this}}function A(){r.e(439).then(r.bind(r,7438)).then((t=>{let{setAPI:r}=t;r(e),(0,c.L)(e,"api")})).catch((()=>(0,f.Z)("Downloading runtime APIs failed...")))}return["actionText","setName","setAttribute","save","ignore","onEnd","getContext","end","get"].forEach((e=>{w[e]=x(v,e,void 0,n.D.spa)})),h.noticeError=function(e,t){"string"==typeof e&&(e=new Error(e)),(0,o.p)(l.xS,["API/noticeError/called"],void 0,n.D.metrics,g),(0,o.p)("err",[e,(0,s.z)(),!1,t],void 0,n.D.jserrors,g)},d.il?(0,u.b)((()=>A()),!0):A(),h}(e,v);return(0,h.Qy)(e,T,"api"),(0,h.Qy)(e,A,"exposed"),(0,h.EZ)("activatedFeatures",p.T),T}},3325:(e,t,r)=>{r.d(t,{D:()=>n,p:()=>i});const n={ajax:"ajax",jserrors:"jserrors",metrics:"metrics",pageAction:"page_action",pageViewEvent:"page_view_event",pageViewTiming:"page_view_timing",sessionReplay:"session_replay",sessionTrace:"session_trace",spa:"spa"},i={[n.pageViewEvent]:1,[n.pageViewTiming]:2,[n.metrics]:3,[n.jserrors]:4,[n.ajax]:5,[n.sessionTrace]:6,[n.pageAction]:7,[n.spa]:8,[n.sessionReplay]:9}}},n={};function i(e){var t=n[e];if(void 0!==t)return t.exports;var o=n[e]={exports:{}};return r[e](o,o.exports,i),o.exports}i.m=r,i.d=(e,t)=>{for(var r in t)i.o(t,r)&&!i.o(e,r)&&Object.defineProperty(e,r,{enumerable:!0,get:t[r]})},i.f={},i.e=e=>Promise.all(Object.keys(i.f).reduce(((t,r)=>(i.f[r](e,t),t)),[])),i.u=e=>(({78:"page_action-aggregate",147:"metrics-aggregate",242:"session-manager",317:"jserrors-aggregate",348:"page_view_timing-aggregate",412:"lazy-feature-loader",439:"async-api",538:"recorder",590:"session_replay-aggregate",675:"compressor",733:"session_trace-aggregate",786:"page_view_event-aggregate",873:"spa-aggregate",898:"ajax-aggregate"}[e]||e)+"."+{78:"ac76d497",147:"3dc53903",148:"1a20d5fe",242:"2a64278a",317:"49e41428",348:"bd6de33a",412:"2f55ce66",439:"30bd804e",538:"1b18459f",590:"cf0efb30",675:"ae9f91a8",733:"83105561",786:"06482edd",860:"03a8b7a5",873:"e6b09d52",898:"998ef92b"}[e]+"-1.236.0.min.js"),i.o=(e,t)=>Object.prototype.hasOwnProperty.call(e,t),e={},t="NRBA:",i.l=(r,n,o,a)=>{if(e[r])e[r].push(n);else{var s,c;if(void 0!==o)for(var u=document.getElementsByTagName("script"),d=0;d {s.onerror=s.onload=null,clearTimeout(h);var i=e[r];if(delete e[r],s.parentNode&&s.parentNode.removeChild(s),i&&i.forEach((e=>e(n))),t)return t(n)},h=setTimeout(l.bind(null,void 0,{type:"timeout",target:s}),12e4);s.onerror=l.bind(null,s.onerror),s.onload=l.bind(null,s.onload),c&&document.head.appendChild(s)}},i.r=e=>{"undefined"!=typeof Symbol&&Symbol.toStringTag&&Object.defineProperty(e,Symbol.toStringTag,{value:"Module"}),Object.defineProperty(e,"__esModule",{value:!0})},i.j=364,i.p="https://js-agent.newrelic.com/",(()=>{var e={364:0,953:0};i.f.j=(t,r)=>{var n=i.o(e,t)?e[t]:void 0;if(0!==n)if(n)r.push(n[2]);else{var o=new Promise(((r,i)=>n=e[t]=[r,i]));r.push(n[2]=o);var a=i.p+i.u(t),s=new Error;i.l(a,(r=>{if(i.o(e,t)&&(0!==(n=e[t])&&(e[t]=void 0),n)){var o=r&&("load"===r.type?"missing":r.type),a=r&&r.target&&r.target.src;s.message="Loading chunk "+t+" failed.\n("+o+": "+a+")",s.name="ChunkLoadError",s.type=o,s.request=a,n[1](s)}}),"chunk-"+t,t)}};var t=(t,r)=>{var n,o,[a,s,c]=r,u=0;if(a.some((t=>0!==e[t]))){for(n in s)i.o(s,n)&&(i.m[n]=s[n]);if(c)c(i)}for(t&&t(r);u {i.r(o);var e=i(3325),t=i(5763);const r=Object.values(e.D);function n(e){const n={};return r.forEach((r=>{n[r]=function(e,r){return!1!==(0,t.Mt)(r,"".concat(e,".enabled"))}(r,e)})),n}var a=i(9144);var s=i(5546),c=i(385),u=i(8e3),d=i(5938),f=i(3960),l=i(50);class h extends d.W{constructor(e,t,r){let n=!(arguments.length>3&&void 0!==arguments[3])||arguments[3];super(e,t,r),this.auto=n,this.abortHandler,this.featAggregate,this.onAggregateImported,n&&(0,u.R)(e,r)}importAggregator(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:{};if(this.featAggregate||!this.auto)return;const r=c.il&&!0===(0,t.Mt)(this.agentIdentifier,"privacy.cookies_enabled");let n;this.onAggregateImported=new Promise((e=>{n=e}));const o=async()=>{let t;try{if(r){const{setupAgentSession:e}=await Promise.all([i.e(860),i.e(242)]).then(i.bind(i,3228));t=e(this.agentIdentifier)}}catch(e){(0,l.Z)("A problem occurred when starting up session manager. This page will not start or extend any session.",e)}try{if(!this.shouldImportAgg(this.featureName,t))return void(0,u.L)(this.agentIdentifier,this.featureName);const{lazyFeatureLoader:r}=await i.e(412).then(i.bind(i,8582)),{Aggregate:o}=await r(this.featureName,"aggregate");this.featAggregate=new o(this.agentIdentifier,this.aggregator,e),n(!0)}catch(e){(0,l.Z)("Downloading and initializing ".concat(this.featureName," failed..."),e),this.abortHandler?.(),n(!1)}};c.il?(0,f.b)((()=>o()),!0):o()}shouldImportAgg(r,n){return r!==e.D.sessionReplay||!1!==(0,t.Mt)(this.agentIdentifier,"session_trace.enabled")&&(!!n?.isNew||!!n?.state.sessionReplay)}}var g=i(7633),p=i(7894);class m extends h{static featureName=g.t9;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];if(super(r,n,g.t9,i),("undefined"==typeof PerformanceNavigationTiming||c.Tt)&&"undefined"!=typeof PerformanceTiming){const n=(0,t.OP)(r);n[g.Dz]=Math.max(Date.now()-n.offset,0),(0,f.K)((()=>n[g.qw]=Math.max((0,p.z)()-n[g.Dz],0))),(0,f.b)((()=>{const t=(0,p.z)();n[g.OJ]=Math.max(t-n[g.Dz],0),(0,s.p)("timing",["load",t],void 0,e.D.pageViewTiming,this.ee)}))}this.importAggregator()}}var v=i(1117),b=i(1284);class y extends v.w{constructor(e){super(e),this.aggregatedData={}}store(e,t,r,n,i){var o=this.getBucket(e,t,r,i);return o.metrics=function(e,t){t||(t={count:0});return t.count+=1,(0,b.D)(e,(function(e,r){t[e]=w(r,t[e])})),t}(n,o.metrics),o}merge(e,t,r,n,i){var o=this.getBucket(e,t,n,i);if(o.metrics){var a=o.metrics;a.count+=r.count,(0,b.D)(r,(function(e,t){if("count"!==e){var n=a[e],i=r[e];i&&!i.c?a[e]=w(i.t,n):a[e]=function(e,t){if(!t)return e;t.c||(t=x(t.t));return t.min=Math.min(e.min,t.min),t.max=Math.max(e.max,t.max),t.t+=e.t,t.sos+=e.sos,t.c+=e.c,t}(i,a[e])}}))}else o.metrics=r}storeMetric(e,t,r,n){var i=this.getBucket(e,t,r);return i.stats=w(n,i.stats),i}getBucket(e,t,r,n){this.aggregatedData[e]||(this.aggregatedData[e]={});var i=this.aggregatedData[e][t];return i||(i=this.aggregatedData[e][t]={params:r||{}},n&&(i.custom=n)),i}get(e,t){return t?this.aggregatedData[e]&&this.aggregatedData[e][t]:this.aggregatedData[e]}take(e){for(var t={},r="",n=!1,i=0;i t.max&&(t.max=e),e 2&&void 0!==arguments[2])||arguments[2];super(e,r,j.t,n),c.il&&((0,t.OP)(e).initHidden=Boolean("hidden"===document.visibilityState),(0,N.N)((()=>(0,s.p)("docHidden",[(0,p.z)()],void 0,j.t,this.ee)),!0),(0,O.bP)("pagehide",(()=>(0,s.p)("winPagehide",[(0,p.z)()],void 0,j.t,this.ee))),this.importAggregator())}}var P=i(3081);class C extends h{static featureName=P.t9;constructor(e,t){let r=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(e,t,P.t9,r),this.importAggregator()}}var R,I=i(2210),k=i(1214),H=i(2177),L={};try{R=localStorage.getItem("__nr_flags").split(","),console&&"function"==typeof console.log&&(L.console=!0,-1!==R.indexOf("dev")&&(L.dev=!0),-1!==R.indexOf("nr_dev")&&(L.nrDev=!0))}catch(e){}function z(e){try{L.console&&z(e)}catch(e){}}L.nrDev&&H.ee.on("internal-error",(function(e){z(e.stack)})),L.dev&&H.ee.on("fn-err",(function(e,t,r){z(r.stack)})),L.dev&&(z("NR AGENT IN DEVELOPMENT MODE"),z("flags: "+(0,b.D)(L,(function(e,t){return e})).join(", ")));var M=i(6660);class B extends h{static featureName=M.t;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(r,n,M.t,i),this.skipNext=0;try{this.removeOnAbort=new AbortController}catch(e){}const o=this;o.ee.on("fn-start",(function(e,t,r){o.abortHandler&&(o.skipNext+=1)})),o.ee.on("fn-err",(function(t,r,n){o.abortHandler&&!n[M.A]&&((0,I.X)(n,M.A,(function(){return!0})),this.thrown=!0,(0,s.p)("err",[n,(0,p.z)()],void 0,e.D.jserrors,o.ee))})),o.ee.on("fn-end",(function(){o.abortHandler&&!this.thrown&&o.skipNext>0&&(o.skipNext-=1)})),o.ee.on("internal-error",(function(t){(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,o.ee)})),this.origOnerror=c._A.onerror,c._A.onerror=this.onerrorHandler.bind(this),c._A.addEventListener("unhandledrejection",(t=>{const r=function(e){let t="Unhandled Promise Rejection: ";if(e instanceof Error)try{return e.message=t+e.message,e}catch(t){return e}if(void 0===e)return new Error(t);try{return new Error(t+(0,D.P)(e))}catch(e){return new Error(t)}}(t.reason);(0,s.p)("err",[r,(0,p.z)(),!1,{unhandledPromiseRejection:1}],void 0,e.D.jserrors,this.ee)}),(0,O.m$)(!1,this.removeOnAbort?.signal)),(0,k.gy)(this.ee),(0,k.BV)(this.ee),(0,k.em)(this.ee),(0,t.OP)(r).xhrWrappable&&(0,k.Kf)(this.ee),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}onerrorHandler(t,r,n,i,o){"function"==typeof this.origOnerror&&this.origOnerror(...arguments);try{this.skipNext?this.skipNext-=1:(0,s.p)("err",[o||new F(t,r,n),(0,p.z)()],void 0,e.D.jserrors,this.ee)}catch(t){try{(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,this.ee)}catch(e){}}return!1}}function F(e,t,r){this.message=e||"Uncaught error with no additional information",this.sourceURL=t,this.line=r}let U=1;const q="nr@id";function G(e){const t=typeof e;return!e||"object"!==t&&"function"!==t?-1:e===c._A?0:(0,I.X)(e,q,(function(){return U++}))}function V(e){if("string"==typeof e&&e.length)return e.length;if("object"==typeof e){if("undefined"!=typeof ArrayBuffer&&e instanceof ArrayBuffer&&e.byteLength)return e.byteLength;if("undefined"!=typeof Blob&&e instanceof Blob&&e.size)return e.size;if(!("undefined"!=typeof FormData&&e instanceof FormData))try{return(0,D.P)(e).length}catch(e){return}}}var X=i(7243);class W{constructor(e){this.agentIdentifier=e,this.generateTracePayload=this.generateTracePayload.bind(this),this.shouldGenerateTrace=this.shouldGenerateTrace.bind(this)}generateTracePayload(e){if(!this.shouldGenerateTrace(e))return null;var r=(0,t.DL)(this.agentIdentifier);if(!r)return null;var n=(r.accountID||"").toString()||null,i=(r.agentID||"").toString()||null,o=(r.trustKey||"").toString()||null;if(!n||!i)return null;var a=(0,_.M)(),s=(0,_.Ht)(),c=Date.now(),u={spanId:a,traceId:s,timestamp:c};return(e.sameOrigin||this.isAllowedOrigin(e)&&this.useTraceContextHeadersForCors())&&(u.traceContextParentHeader=this.generateTraceContextParentHeader(a,s),u.traceContextStateHeader=this.generateTraceContextStateHeader(a,c,n,i,o)),(e.sameOrigin&&!this.excludeNewrelicHeader()||!e.sameOrigin&&this.isAllowedOrigin(e)&&this.useNewrelicHeaderForCors())&&(u.newrelicHeader=this.generateTraceHeader(a,s,c,n,i,o)),u}generateTraceContextParentHeader(e,t){return"00-"+t+"-"+e+"-01"}generateTraceContextStateHeader(e,t,r,n,i){return i+"@nr=0-1-"+r+"-"+n+"-"+e+"----"+t}generateTraceHeader(e,t,r,n,i,o){if(!("function"==typeof c._A?.btoa))return null;var a={v:[0,1],d:{ty:"Browser",ac:n,ap:i,id:e,tr:t,ti:r}};return o&&n!==o&&(a.d.tk=o),btoa((0,D.P)(a))}shouldGenerateTrace(e){return this.isDtEnabled()&&this.isAllowedOrigin(e)}isAllowedOrigin(e){var r=!1,n={};if((0,t.Mt)(this.agentIdentifier,"distributed_tracing")&&(n=(0,t.P_)(this.agentIdentifier).distributed_tracing),e.sameOrigin)r=!0;else if(n.allowed_origins instanceof Array)for(var i=0;i 2&&void 0!==arguments[2])||arguments[2];super(r,n,Z.t,i),(0,t.OP)(r).xhrWrappable&&(this.dt=new W(r),this.handler=(e,t,r,n)=>(0,s.p)(e,t,r,n,this.ee),(0,k.u5)(this.ee),(0,k.Kf)(this.ee),function(r,n,i,o){function a(e){var t=this;t.totalCbs=0,t.called=0,t.cbTime=0,t.end=E,t.ended=!1,t.xhrGuids={},t.lastSize=null,t.loadCaptureCalled=!1,t.params=this.params||{},t.metrics=this.metrics||{},e.addEventListener("load",(function(r){_(t,e)}),(0,O.m$)(!1)),c.IF||e.addEventListener("progress",(function(e){t.lastSize=e.loaded}),(0,O.m$)(!1))}function s(e){this.params={method:e[0]},T(this,e[1]),this.metrics={}}function u(e,n){var i=(0,t.DL)(r);i.xpid&&this.sameOrigin&&n.setRequestHeader("X-NewRelic-ID",i.xpid);var a=o.generateTracePayload(this.parsedOrigin);if(a){var s=!1;a.newrelicHeader&&(n.setRequestHeader("newrelic",a.newrelicHeader),s=!0),a.traceContextParentHeader&&(n.setRequestHeader("traceparent",a.traceContextParentHeader),a.traceContextStateHeader&&n.setRequestHeader("tracestate",a.traceContextStateHeader),s=!0),s&&(this.dt=a)}}function d(e,t){var r=this.metrics,i=e[0],o=this;if(r&&i){var a=V(i);a&&(r.txSize=a)}this.startTime=(0,p.z)(),this.listener=function(e){try{"abort"!==e.type||o.loadCaptureCalled||(o.params.aborted=!0),("load"!==e.type||o.called===o.totalCbs&&(o.onloadCalled||"function"!=typeof t.onload)&&"function"==typeof o.end)&&o.end(t)}catch(e){try{n.emit("internal-error",[e])}catch(e){}}};for(var s=0;s 1?e[1]=i:e.push(i)}else e[0]&&e[0].headers&&s(e[0].headers,n)&&(this.dt=n);function s(e,t){var r=!1;return t.newrelicHeader&&(e.set("newrelic",t.newrelicHeader),r=!0),t.traceContextParentHeader&&(e.set("traceparent",t.traceContextParentHeader),t.traceContextStateHeader&&e.set("tracestate",t.traceContextStateHeader),r=!0),r}}function x(e,t){this.params={},this.metrics={},this.startTime=(0,p.z)(),this.dt=t,e.length>=1&&(this.target=e[0]),e.length>=2&&(this.opts=e[1]);var r,n=this.opts||{},i=this.target;"string"==typeof i?r=i:"object"==typeof i&&i instanceof Y?r=i.url:c._A?.URL&&"object"==typeof i&&i instanceof URL&&(r=i.href),T(this,r);var o=(""+(i&&i instanceof Y&&i.method||n.method||"GET")).toUpperCase();this.params.method=o,this.txSize=V(n.body)||0}function A(t,r){var n;this.endTime=(0,p.z)(),this.params||(this.params={}),this.params.status=r?r.status:0,"string"==typeof this.rxSize&&this.rxSize.length>0&&(n=+this.rxSize);var o={txSize:this.txSize,rxSize:n,duration:(0,p.z)()-this.startTime};i("xhr",[this.params,o,this.startTime,this.endTime,"fetch"],this,e.D.ajax)}function E(t){var r=this.params,n=this.metrics;if(!this.ended){this.ended=!0;for(var o=0;o 2&&void 0!==arguments[2])||arguments[2];super(e,t,we.t,r),this.importAggregator()}}new class{constructor(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:(0,_.ky)(16);c._A?(this.agentIdentifier=t,this.sharedAggregator=new y({agentIdentifier:this.agentIdentifier}),this.features={},this.desiredFeatures=new Set(e.features||[]),this.desiredFeatures.add(m),Object.assign(this,(0,a.j)(this.agentIdentifier,e,e.loaderType||"agent")),this.start()):(0,l.Z)("Failed to initial the agent. Could not determine the runtime environment.")}get config(){return{info:(0,t.C5)(this.agentIdentifier),init:(0,t.P_)(this.agentIdentifier),loader_config:(0,t.DL)(this.agentIdentifier),runtime:(0,t.OP)(this.agentIdentifier)}}start(){const t="features";try{const r=n(this.agentIdentifier),i=[...this.desiredFeatures];i.sort(((t,r)=>e.p[t.featureName]-e.p[r.featureName])),i.forEach((t=>{if(r[t.featureName]||t.featureName===e.D.pageViewEvent){const n=function(t){switch(t){case e.D.ajax:return[e.D.jserrors];case e.D.sessionTrace:return[e.D.ajax,e.D.pageViewEvent];case e.D.sessionReplay:return[e.D.sessionTrace];case e.D.pageViewTiming:return[e.D.pageViewEvent];default:return[]}}(t.featureName);n.every((e=>r[e]))||(0,l.Z)("".concat(t.featureName," is enabled but one or more dependent features has been disabled (").concat((0,D.P)(n),"). This may cause unintended consequences or missing data...")),this.features[t.featureName]=new t(this.agentIdentifier,this.sharedAggregator)}})),(0,T.Qy)(this.agentIdentifier,this.features,t)}catch(e){(0,l.Z)("Failed to initialize all enabled instrument classes (agent aborted) -",e);for(const e in this.features)this.features[e].abortHandler?.();const r=(0,T.fP)();return delete r.initializedAgents[this.agentIdentifier]?.api,delete r.initializedAgents[this.agentIdentifier]?.[t],delete this.sharedAggregator,r.ee?.abort(),delete r.ee?.get(this.agentIdentifier),!1}}}({features:[J,m,S,class extends h{static featureName=oe;constructor(t,r){if(super(t,r,oe,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;const n=this.ee;let i;(0,k.QU)(n),this.eventsEE=(0,k.em)(n),this.eventsEE.on(se,(function(e,t){this.bstStart=(0,p.z)()})),this.eventsEE.on(ae,(function(t,r){(0,s.p)("bst",[t[0],r,this.bstStart,(0,p.z)()],void 0,e.D.sessionTrace,n)})),n.on(ce+ne,(function(e){this.time=(0,p.z)(),this.startPath=location.pathname+location.hash})),n.on(ce+ie,(function(t){(0,s.p)("bstHist",[location.pathname+location.hash,this.startPath,this.time],void 0,e.D.sessionTrace,n)}));try{i=new PerformanceObserver((t=>{const r=t.getEntries();(0,s.p)(te,[r],void 0,e.D.sessionTrace,n)})),i.observe({type:re,buffered:!0})}catch(e){}this.importAggregator({resourceObserver:i})}},C,xe,B,class extends h{static featureName=de;constructor(e,r){if(super(e,r,de,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;if(!(0,t.OP)(e).xhrWrappable)return;try{this.removeOnAbort=new AbortController}catch(e){}let n,i=0;const o=this.ee.get("tracer"),a=(0,k._L)(this.ee),s=(0,k.Lg)(this.ee),u=(0,k.BV)(this.ee),d=(0,k.Kf)(this.ee),f=this.ee.get("events"),l=(0,k.u5)(this.ee),h=(0,k.QU)(this.ee),g=(0,k.Gm)(this.ee);function m(e,t){h.emit("newURL",[""+window.location,t])}function v(){i++,n=window.location.hash,this[ve]=(0,p.z)()}function b(){i--,window.location.hash!==n&&m(0,!0);var e=(0,p.z)();this[pe]=~~this[pe]+e-this[ve],this[ye]=e}function y(e,t){e.on(t,(function(){this[t]=(0,p.z)()}))}this.ee.on(ve,v),s.on(be,v),a.on(be,v),this.ee.on(ye,b),s.on(ge,b),a.on(ge,b),this.ee.buffer([ve,ye,"xhr-resolved"],this.featureName),f.buffer([ve],this.featureName),u.buffer(["setTimeout"+le,"clearTimeout"+fe,ve],this.featureName),d.buffer([ve,"new-xhr","send-xhr"+fe],this.featureName),l.buffer([me+fe,me+"-done",me+he+fe,me+he+le],this.featureName),h.buffer(["newURL"],this.featureName),g.buffer([ve],this.featureName),s.buffer(["propagate",be,ge,"executor-err","resolve"+fe],this.featureName),o.buffer([ve,"no-"+ve],this.featureName),a.buffer(["new-jsonp","cb-start","jsonp-error","jsonp-end"],this.featureName),y(l,me+fe),y(l,me+"-done"),y(a,"new-jsonp"),y(a,"jsonp-end"),y(a,"cb-start"),h.on("pushState-end",m),h.on("replaceState-end",m),window.addEventListener("hashchange",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("load",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("popstate",(function(){m(0,i>1)}),(0,O.m$)(!0,this.removeOnAbort?.signal)),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}}],loaderType:"spa"})})(),window.NRBA=o})(); window.jQuery || document.write(' ') CKEDITOR_BASEPATH='https://f1000research.com/js/vendor/ckeditor/' window.reactTheme = 'research'; window.MathJax = { CommonHTML: { linebreaks: { automatic: true } }, 'HTML-CSS': { linebreaks: { automatic: true } }, SVG: { linebreaks: { automatic: true } }, AuthorInit: function() { MathJax.Hub.Register.MessageHook('End Process', function () { let timeout = false; // holder for timeout id const delay = 250; // delay after event is "complete" to run callback const reflowMath = function() { const dispFormulas = document.querySelectorAll('.disp-formula.panel'); if (!dispFormulas) { return; } for (const dispFormula of dispFormulas) { const child = dispFormula.querySelector('.MathJax_Preview').nextSibling.firstChild; const isMultiline = MathJax.Hub.getAllJax(dispFormula)[0].root.isMultiline; if (dispFormula.offsetWidth < child.offsetWidth || isMultiline) { MathJax.Hub.Queue(['Rerender', MathJax.Hub, dispFormula]); } } }; window.addEventListener('resize', function() { clearTimeout(timeout); // clear the timeout timeout = setTimeout(reflowMath, delay); // start timing for event "completion" }); }); }, }; if (window.location.hash == '#_=_'){ window.location = window.location.href.split('#')[0] } !function(f,b,e,v,n,t,s){if(f.fbq)return;n=f.fbq=function() {n.callMethod? n.callMethod.apply(n,arguments):n.queue.push(arguments)} ;if(!f._fbq)f._fbq=n; n.push=n;n.loaded=!0;n.version='2.0';n.queue=[];t=b.createElement(e);t.async=!0; t.src=v;s=b.getElementsByTagName(e)[0];s.parentNode.insertBefore(t,s)}(window, document,'script','https://connect.facebook.net/en_US/fbevents.js'); fbq('init', '1641728616063202'); fbq('track', "PixelInitialized", {}); (function(h,o,t,j,a,r){ h.hj=h.hj||function(){(h.hj.q=h.hj.q||[]).push(arguments)}; h._hjSettings={hjid:2318163,hjsv:6}; a=o.getElementsByTagName('head')[0]; r=o.createElement('script');r.async=1; r.src=t+h._hjSettings.hjid+j+h._hjSettings.hjsv; a.appendChild(r); })(window,document,'https://static.hotjar.com/c/hotjar-','.js?sv='); search file_upload Submit your research search menu close search Browse Gateways & Collections How to Publish Submit your Research My Submissions Article Guidelines Article Guidelines (New Versions) Open Data, Software and Code Guidelines Open Data and Accessible Source Materials Guidelines (HSS) Open Data, Software and Code Guidelines (PSE) Prepublication Checks Production Process Posters and Slides Guidelines Document Guidelines Article Processing Charges Peer Review Finding Article Reviewers About How it Works For Reviewers Our Advisors Policies Glossary FAQs For Developers Newsroom Contact My Research Submissions Content and Tracking Alerts My Details Sign In file_upload Submit your research { "@context": "https://schema.org", "@type": "ScholarlyArticle", "mainEntityOfPage": { "@type": "WebPage", "@id": "https://f1000research.com/articles/14-694" }, "headline": "Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point", "datePublished": "2025-07-16T15:16:16", "dateModified": "2026-01-16T12:25:26", "author": [ { "@type": "Person", "name": "Mohammed Q. Shormani" }, { "@type": "Person", "name": "Alia. Ali Al-Samki" } ], "publisher": { "@type": "Organization", "name": "F1000Research", "logo": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 480, "width": 60 } }, "image": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 1200, "width": 150 }, "description": " Background This study aims to detect the efficiency of two Artificial Intelligence (AI) translation models, ChatGPT and DeepSeek, in the translation of Yemeni San’ani Arabic (YSA) dialectical terms into English. As dialectal Arabic presents significant linguistic variability and cultural specificity, accurate translation remains a major challenge for the current ChatGPT and DeepSeek (and perhaps other AI models). Methods Fifty San’ani Arabic terms were involved in the translation process, assessing the ability of both models to capture their semantic fidelity, cultural relevance, and contextual accuracy. Results The study findings reveal that, while both models demonstrate a foundational understanding of Standard Arabic (SA), their performance diminishes considerably when faced with the nuances and idiomatic expressions of the San’ani Arabic dialect. ChatGPT displays a relatively better performance in certain cases, particularly when translating terms with dialectical connotations. However, both models exhibit limitations, such as literal translation, misinterpretation, or complete ignorance of the intended meaning. Conclusions The study concludes by highlighting the critical need for dialect-aware AI development and provides recommendations for improving the dialectical accuracy and cultural sensitivity of AI model translation. " } { "@context": "http://schema.org", "@type": "BreadcrumbList", "itemListElement": [ { "@type": "ListItem", "position": "1", "item": { "@id": "https://f1000research.com/", "name": "Home" } }, { "@type": "ListItem", "position": "2", "item": { "@id": "https://f1000research.com/browse/articles", "name": "Browse" } }, { "@type": "ListItem", "position": "3", "item": { "@id": "https://f1000research.com/articles/14-694/v2", "name": "Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani..." } } ] } Home Browse Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani... ALL Metrics - Views Downloads Get PDF Get XML Cite How to cite this article Shormani MQ and Al-Samki AA. Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.12688/f1000research.165879.2 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. Close Copy Citation Details Export Export Citation Sciwheel EndNote Ref. Manager Bibtex ProCite Sente EXPORT Select a format first Track Share ▬ ✚ Research Article Revised Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] Mohammed Q. Shormani 1 , Alia. Ali Al-Samki 1 Mohammed Q. Shormani 1 , Alia. Ali Al-Samki 1 PUBLISHED 16 Jan 2026 Author details Author details 1 Department of English Studies, Ibb University, Ibb, Ibb Governorate, Yemen Mohammed Q. Shormani Roles: Conceptualization, Formal Analysis, Methodology, Software, Writing – Original Draft Preparation, Writing – Review & Editing Alia. Ali Al-Samki Roles: Data Curation, Resources, Validation, Writing – Original Draft Preparation, Writing – Review & Editing OPEN PEER REVIEW DETAILS REVIEWER STATUS This article is included in the Artificial Intelligence and Machine Learning gateway. Abstract Background This study aims to detect the efficiency of two Artificial Intelligence (AI) translation models, ChatGPT and DeepSeek, in the translation of Yemeni San’ani Arabic (YSA) dialectical terms into English. As dialectal Arabic presents significant linguistic variability and cultural specificity, accurate translation remains a major challenge for the current ChatGPT and DeepSeek (and perhaps other AI models). Methods Fifty San’ani Arabic terms were involved in the translation process, assessing the ability of both models to capture their semantic fidelity, cultural relevance, and contextual accuracy. Results The study findings reveal that, while both models demonstrate a foundational understanding of Standard Arabic (SA), their performance diminishes considerably when faced with the nuances and idiomatic expressions of the San’ani Arabic dialect. ChatGPT displays a relatively better performance in certain cases, particularly when translating terms with dialectical connotations. However, both models exhibit limitations, such as literal translation, misinterpretation, or complete ignorance of the intended meaning. Conclusions The study concludes by highlighting the critical need for dialect-aware AI development and provides recommendations for improving the dialectical accuracy and cultural sensitivity of AI model translation. READ ALL READ LESS Keywords Dialect translation, ChatGPT, DeepSeek, Sana’ani Arabic, dialectical and cultural nuances Corresponding Author(s) Mohammed Q. Shormani ( [email protected] ) Close Corresponding author: Mohammed Q. Shormani Competing interests: No competing interests were disclosed. Grant information: The author(s) declared that no grants were involved in supporting this work. Copyright: © 2026 Shormani MQ and Al-Samki AA. This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. How to cite: Shormani MQ and Al-Samki AA. Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.12688/f1000research.165879.2 ) First published: 16 Jul 2025, 14 :694 ( https://doi.org/10.12688/f1000research.165879.1 ) Latest published: 16 Jan 2026, 14 :694 ( https://doi.org/10.12688/f1000research.165879.2 ) Revised Amendments from Version 1 This revised version of the article substantially extends and refines the previously published study in terms of theoretical grounding, methodological rigor, and analytical depth. The literature review has been extended to focus more explicitly on Arabic dialect machine translation, and updated references on DeepSeek. Methodologically, the study now provides a more transparent account of data selection, criteria, evaluation criteria, and reproducibility, including clearer definitions of (in)correctness and appropriateness, specific recommendations for AI developers concerning YSA, expanding the limitations to include those concerning utilizing isolated words in the study, and suggesting further research to address these limitations. This revised version of the article substantially extends and refines the previously published study in terms of theoretical grounding, methodological rigor, and analytical depth. The literature review has been extended to focus more explicitly on Arabic dialect machine translation, and updated references on DeepSeek. Methodologically, the study now provides a more transparent account of data selection, criteria, evaluation criteria, and reproducibility, including clearer definitions of (in)correctness and appropriateness, specific recommendations for AI developers concerning YSA, expanding the limitations to include those concerning utilizing isolated words in the study, and suggesting further research to address these limitations. See the authors' detailed response to the review by Liang Ding See the authors' detailed response to the review by Cao-Tuong DINH READ REVIEWER RESPONSES 1. Introduction In the era of global communication, technological and digital advancements, and for the sake of ease and building varied cultural relationships among nations of different languages, conventional manual translation has been revolutionized by the integration of automated machine translation artificial intelligence (AI) tools, which have largely taken a significant part in translation procedures ( Alafnan, 2024 ; Koehn, 2009 ; Ali et al., 2023 ; Çetin & Duran, 2024 ; Puppel & Borg, 2024 ). Several AI electronic tools have been employed in the execution of translation, some of the most prominent of which are ChatGPT, DeepSeek, DeepL, Google Translate, Babylon, Amazon Translate, Yandex Translate, Systran, and Bing Microsoft Translator. There is a huge argument regarding the efficient use of such tools and their potential to replace the roles of humans in the process of translation, and millions use them daily without scrutinization and with no evaluation ( Çetin & Duran, 2024 ; Postigo, 2024 ). The output of such intelligent tools, however, is not error-free; and thus, many researches have studied the effectiveness of such tools not for getting perfect translation models but for minimizing the rate of inadequacy in translation outcomes and enhancing their outputs. As an elaboration of such studies, this study focuses on the efficiency of two notable AI models, viz. ChatGPT and DeepSeek (cf. Çetin & Duran, 2024 ), and compares their performances in translating 50 San’ani dialectical terms since dialects are considered to be of great importance in linguistic studies and culturally global communication ( Bamunusinghe & Bamunusinghe, 2014 ). The insights of this study are used to identify their strengths and limitations. They are expected to be an appropriate overview used for optimizing the efficiency of AI translation tools by their designers as well as human translators. Simply put, this study aims to evaluate and compare the efficiency of ChatGPT and DeepSeek AI tools in the translation of San’ani dialectical terms into English. It also aims to identify their strengths and weaknesses, hoping to optimize the efficiency of AI translation tools by their designers and human translators. Thus, the significance of this study lies in evaluating and comparing the efficiency of ChatGPT and DeepSeek in the translation of San’ani dialectical terms into English. To the best of our knowledge, this is the first academic study to address the efficiency of DeepSeek model and compare it to ChatGPT in terms of the San’ani dialect. San’ani Arabic is spoken in Sana’a governorate, the capital of Yemen. Like other Yemeni Arabic varieties, San’ani Arabic is an understudied language variety that necessitates and urges linguists to address its distinct linguistic features (cf. Shormani, 2019 ). The remainder of this paper is organized as follows. In Section 2 , the study posits the conceptual framework, tackling a historical overview of machine translation and two of its currently notable AI models. Section 3 outlines the translating of dialects and focuses on related studies. Section 4 tackles the study methodology. Section 5 analyzes and discuss the results, respectively. Section 6 presents the study conclusions, recommendations, and limitations. 2. Conceptual framework 2.1 Machine translation systems The implementation of technology in translation has a long history; however, it has become highly remarkable in recent decades ( Postigo, 2024 ). Machine Translation (MT) systems “have been at the forefront of translation technology since the 1950s” and has crossed through major developments ( Postigo, 2024 : 1; Alafnan, 2024 ; Çetin & Duran, 2024 ) and by the time, they got more directed towards better mirroring of natural language processing representing a surprising shift in translation technology ( Koehn, 2009 ; Postigo, 2024 ). MT is “one of the applications studied in computational linguistics” ( Alafnan, 2024 , p. 21). The processes are based on codes, encoding, and decoding. MT has three main approaches categorized according to its functionality: Rule-based Machine Translation (RMT), which extended from the nearly 1960s to the early 1990s; Statistical Machine Translation (SMT), which extended from the early 1990s to the 2010s; and Neural Machine Translation (NMT), which was shaped nearly in 2010 and extended up to now ( Hutchins, 1986 ; Koehn et al., 2003 ; Hofmann et al., 2010 ; Wu et al., 2016 ; Castellani, 2017 ; Vaswani et al. 2017 ; Elkaffash, 2020 ; Shormani, 2024a & b ; Alafnan, 2024 ; Çetin & Duran, 2024 ; Postigo, 2024 ; Sindhuja, 2021 ; Shormani & Alfahd, 2025 ). The first is based on predefined forms, rules, and lexemes posited by expert linguists, according to which translations are generated. It revolves around a linguistic equation of word forms, and phrase structure rules in the source language (SL) that can be rendered into counterpart word forms and phrase structures in the target language (TL). This approach is best represented by Systran and is suitable for translating simple texts rather than complex linguistic constructions and idioms ( Shormani & AlSohbani, 2025 ). The second approach is based on statistics of phrase-strings corpora rather than just word forms, whereby a large amount of data and texts in two languages are analyzed in alternative forms, patterned, and best modeled for usage in fore-coming texts that need to be translated ( Shormani & AlSohbani, 2025 ). However, it still has grammatical and semantic weaknesses, as well as contextual limitations. The third approach, which is the most widely used nowadays, is distinguished from the other two approaches in that it uses deep neural networks and is more contextualized. That is, when translating a text, it looks at the context of the text as a whole, providing better translation than RMT and SMT ( Shormani & AlSohbani, 2025 ). The stages of the three approaches are discussed further in the following sections. Rule-based machine translation was the earliest approach to MT, which emerged in the 1960s as a solution to linguistic barriers in cross-border communication. Initially, the RMT was developed to assist the US Air Forces in translating Russian documents during the Cold War, aiming to facilitate intelligence gathering, diplomatic efforts, and communication among people ( Hutchins, 1986 ). The term “rule-based” refers to the application of grammatical rules, syntactic parsing, and predefined linguistic structures to convert text from one language to another (see e.g., Hutchins, 1986 ; Koehn et al., 2003 ; Shormani & AlSohbani, 2025 ). RMT relies on explicit human-crafted rules rather than learning from data. Linguists and computational experts manually defined syntax, morphology, and semantics for both source and target languages, creating structured frameworks for MT (cf. Shormani & AlSohbani, 2025 ; Shormani, 2024b ). This approach requires extensive linguistic expertise, making it both resource-intensive and time-consuming. However, the deterministic nature of RMT ensures predictable translations for specific language pairs, making it a valuable tool for structured and well-defined text types such as legal and technical documents ( Hutchins, 1986 ). RMT systems performed well when translating from and into languages with similar syntactic structures or typologically similar languages, as in the case of English and German, where word order and syntax followed relatively similar patterns. However, these tools struggle with languages that have significant structural differences, such as English and Korean, owing to variations in sentence formation and morphological complexity (see also Castellani, 2017 ). Additionally, RMT found it challenging to process idiomatic and proverbial expressions (cf. Shormani, 2020 ), which often do not have direct equivalents in the target language. Cultural nuances, figurative language, and polysemic words (i.e., words with multiple meanings) pose major obstacles, leading to awkward or inaccurate translations. For example, the English idiom kick the bucket , meaning to die , would be translated into Arabic by RMT literally as يركل الدلو which is far away from “to die,” thus losing its intended meaning in Arabic. Another structure causing difficulty for RMT involves center-embedding sentence structures, where clauses are nested within other clauses, resulting in a complex sentence (cf. Shormani, 2025a ). This nesting often results in poor translations due to the difficulty in processing these structures by RMT. These issues highlight the limitations of a purely rule-based approach and demonstrate the need for more flexible translation methods. Another major problem with RMT is its scalability. Because linguistic rules had to be manually crafted and refined for each language pair, developing RMT for multiple languages was both effort- and time-consuming, and costly. Maintaining and updating rule sets requires continuous input from expert linguists, which makes it difficult to adapt to new linguistic changes or variations (see also Castellani, 2017 ; Elkaffash, 2020 ). Additionally, RMT is unable to effectively handle the vast diversity of natural language, as language evolves with time, context, and usage. Despite these limitations, RMT has laid the groundwork for future advancements in machine translation by emphasizing the importance of linguistic structure. Recognizing the inefficiencies of RMT, researchers have sought alternative approaches that can handle translation tasks more dynamically and efficiently ( Brown et al., 1993 ; Elkaffash, 2020 ; Shormani & AlSohbani, 2025 ). This has led to the emergence of Statistical Machine Translation (SMT), which has shifted from manually defined rules to a probabilistic, data-driven approach to translation. Statistical machine translation emerged in the 1990s as a probabilistic alternative to RMT, leveraging large bilingual corpora and statistical models to improve translation accuracy (see e.g., Hutchins, 1986 ; Shormani, 2024a & b ). Unlike RMT, which relies on predefined linguistic rules, SMT introduces a probabilistic approach using vast amounts of parallel texts to predict the most accurate translations ( Brown et al., 1993 ; Elkaffash, 2020 ). At its core, SMT operates by analyzing bilingual corpora, where sentences in one language are aligned with their corresponding translations in another. The system then applies statistical models to determine the probability of a given target language sentence based on source language input. One of the earliest and most influential SMT frameworks was IBM’s model series, which introduced word-alignment techniques and phrase-based translation methods ( Koehn et al., 2003 ). Another major innovation in SMT is the noisy-channel model (see e.g. Neubig et al., 2010 ; Hofmann et al., 2010 ; Saito et al., 2012 ), which treats translation as a decoding process in which the most probable output is selected based on statistical similarity ( Koehn et al., 2003 ). These statistical methods significantly improve translation fluency compared to RMT by allowing the system to adapt dynamically to large datasets. However, SMT faces several challenges, particularly in handling complex syntactic structures and long-range dependencies ( Hofmann et al., 2010 ). Since SMT relies on phrase-level probabilities rather than deep linguistic understanding, it often struggles with sentence coherence and grammatical correctness. Thus, while SMT can accurately translate isolated words or short phrases, it often fails to maintain the logical flow of longer sentences, resulting in disjointed or incoherently structured outputs. Additionally, SMT models require extensive training data, that is, low-resource languages such as Hindi, Marathi, and Irish are often inadequately represented ( Shormani & AlSohbani, 2025 ; Shormani, 2024b & c ). As a result, translations of these languages tended to be less reliable, with significant errors in syntax and semantics. Early versions of Google Translate, which originally relied on SMT ( Elkaffash, 2020 ), demonstrated these weaknesses by frequently generating grammatically inconsistent and contextually inaccurate translations. Although SMT improved translation accuracy compared to RMT, it was still far from achieving human-level accuracy. One of the major limitations of SMT is its dependence on large high-quality bilingual corpora. The availability of these corpora varied greatly between languages, with high-resource languages such as English, Spanish, and French benefiting from better-trained models, while other languages, specifically low-resource languages, remained underrepresented. Moreover, as noted by Elkaffash (2020) , SMT requires significant human interference to refine translations, which is known as the post-editing process (cf. Groves & Dag, 2009 ; Krings, 2001 ; de Almeida & O’Brien, 2010 ). A post-editing process is often necessary to correct errors in grammar and meaning. Recognizing these shortcomings, researchers have sought more advanced techniques that can better capture the nuances of natural language and improve its contextual accuracy. In response to these challenges, neural AI translation was introduced, as we will see below, marking a significant leap forward in machine translation utilizing deep learning and artificial neural networks. NMT aims to overcome the limitations of SMT by modeling entire sentences as continuous representations, thereby allowing for greater contextual understanding and fluency. The NMT revolutionized the field of MT in the 2010s by introducing deep learning and artificial neural networks. Unlike SMT, which translates text at the phrase level using probabilistic models, NMT considers entire sentences and their broader contextual relationship. This approach allows for more fluent, coherent, and natural-sounding translations, reducing the disjointed outputs that are common in SMT. The fundamental shift brought about by NMT was its ability to process words not as isolated units but as part of a continuous representation, using deep learning techniques to capture the semantic and syntactic structures of a sentence. Early NMT models were based on Neural Networking Algorithms (NNAs) and Recurrent Neural Networks (RNNs), which helped maintain sequential dependencies during translation. This is particularly true with the introduction of Transformers and Vectors. However, RNNs struggle with long-range dependencies, meaning that the quality of translation decreases for longer and more complex sentences. This limitation prompted the search for more effective architectures that could process language in a nonsequential, context-aware manner. A major breakthrough came in 2017 when Vaswani and colleagues introduced the transformer architecture, which replaced RNNs in many NMT models, including ChatGPT (see also Lee, 2023 ; Siu, 2023 ; Kumar et al., 2024 ; Shormani, 2024a & b ). Unlike RNNs, transformers process entire sequences of words, phrases, and sentences simultaneously, making them significantly more efficient and capable of handling long-range dependencies ( Vaswani et al., 2017 ). The self-attention mechanism in transformers allows models to weigh the importance of different words in a sentence, ensuring that translations preserve their meaning across languages. This innovation has led to the rise of Large Language Models (LLMs), such as OpenAI’s Generative Pre-trained Transformer (GPT), which leveraged transformers to produce more contextually aware and fluent translations. NMT dramatically improved translation accuracy, minimized common issues such as word-for-word errors, and enhanced contextual understanding. As a result, major translation platforms, including Google Translate and DeepL, have adopted NMT to replace older SMT-based models. The continued refinement of transformer-based architectures has brought machine translation closer to the human-level sound transition output (see also Jiao et al., 2023 ; Lee, 2023 ; Siu, 2023 ; Kumar et al., 2024 ). Another major advancement of the AI translation industry is the incorporation of neural vectors. However, with its advancements, NMT still faces challenges, particularly when dealing with culture- and religion-based texts (see e.g., Shormani & AlSohbani, 2025 ; Shormani, 2025b ). Unlike other types of texts, such as technical and scientific texts, cultural- and religion-based texts require more than simply rendering wordings ( Shormani, 2020 ); they demand a deep understanding of cultural, religious, and historical contexts ( Shormani, 2025b ). Many NMT models, including ChatGPT, struggle to accurately translate idiomatic expressions, culturally specific texts, and religious terminology (cf. Shormani, 2020 ). This is because AI models primarily learn from vast amounts of (Internet) data, which may not include data containing these cultural nuances or the sensitivities of religious discourse. Additionally, ethical concerns arise when translating sensitive materials, as different cultures have varying interpretations of words and concepts ( Shormani & AlSohbani, 2025 ). Addressing these limitations requires further advancements in AI translation training data, methods, dataset diversification, and post-editing to refine NMT’s ability to handle complex cultural and religious texts effectively (see e.g., Shormani, 2024a & b ). ChatGPT was introduced in 2022 by OpenAI. It is based on the syntactic, morphological, logical, and algorithmic transformation of a large number of language samples, patterns, dictionaries, information, and models, known as Large Language Models, through which it is enabled to generate new outputs and more distinctly adhere to contextual backgrounds for further usage when performing new tasks ( Gill & Kaur, 2023 ; Çetin & Duran, 2024 ; Puppel & Borg, 2024 ). Included within Generative Pre-Trained Transformer (GPT) systems, specifically GPT-3.5 and GPT-4, ChatGPT is enabled to generate new outputs and perform many tasks, some of which are transforming data, translating texts, post-editing, translation evaluation, summarizing, and responding to users’ inquiries and conversations ( Gill & Kaur, 2023 ; Jiang, et al., 2024 ; Macken, 2024 ; Puppel & Borg, 2024 ). It can perform various tasks and generate different outcomes, such as written, visual, or auditory, that almost resemble those produced by humans. It essentially benefits from deep learning and Natural Language Processing, which is a branch of AI that teaches machines to comprehend human language and produce similar output models ( Gill & Kaur, 2023 ). As a result of sustainable technological advancements as well as an alternative novel tool competing with other AI tools in the market with its high scalability and efficient performance, accessible open-source DeepSeek emerged in January 2025 ( Guo, et al. 2024 ; Joshi, 2025 ; Peng, et al. 2025 ; Wang & Kantarcioglu, 2025 ). Like ChatGPT, DeepSeek is a transformative, Large Language Model (LLM) that performs various tasks, such as mathematical and structural reasoning, language processing, problem solving, finance, and healthcare diagnosis. It is comprehensively pretrained on high-quality linguistic and syntactic corpora of codes ( Guo, et al. 2024 ). It features the architecture of Mixture of Experts and Multi-Head Latent Innovation ( Joshi, 2025 ; Peng, et al. 2025 ; Wang & Kantarcioglu, 2025 ). DeepSeek, despite having challenges in innovative tasks and the safety of users, has many advantages and an inspiring future that exceeds those of other AI tools, including ChatGPT and Codex ( Guo, et al. 2024 ; Joshi, 2025 ). Some of its major advantages are grammatical accuracy and contextual evaluation ( Joshi, 2025 ). 3. Translating dialects Translating nonstandard dialects presents significant challenges for human translators because of their lack of formal codification, regional variation, and deep cultural embedding. Unlike standardized languages, which have established grammatical rules and extensive documentation, non-standard dialects often rely on oral traditions and are shaped by local customs, idioms, and phonetic shifts that may not have direct equivalents in the target language ( Federici, 2011 ; Kong, 2013 ). Additionally, dialects often carry the connotations of social class, regional identity, and cultural heritage. Translating these aspects requires more than linguistic proficiency; cultural fluency is required to ensure that the translation resonates with the target audience while preserving the source’s authenticity. For example, when translating Swedish dialects into English, preserving local identity and authenticity is crucial, as dialects are expressions of local identity and community ( Federici, 2011 ). Thus, if translating dialects is difficult for human translators, then it is expected that MT tools face more difficulty in this regard because these dialects are not within the training data. Put simply, these tools struggle with non-standard dialects; their training data are typically standard language corpora and may not accurately process vernacular speech, leading to errors and loss of meaning (see also Puppel & Borg, 2024 ). Arabic is a diglossic language with several dialects and vernaculars (see also Ferguson, 1959 ). In this study, we examine this aspect using both ChatGPT and DeepSeek to determine the extent to which they can translate Yemeni San’ani Arabic (YSA). For a decade or so, a number of studies have examined and evaluated the workflow of AI translation tools, seeking to identify their efficiency and usefulness, as well as their weaknesses and limitations in translation. Taking the English-German pair as a case study, Puppel and Borg (2024) evaluated the performance of ChatGPT and showed its strengths and weaknesses based on prompts. The most prominent strength of ChatGPT translation is the coherence of the translated output. However, the prevailing limitation of ChatGPT exhibited in this study is related to style. Other limitations include fluency and accuracy. This study highlighted the need for post-edition of ChatGPT translated output to detect errors and edit them. Therefore, “the intervention of trained translators is required to correct errors and fine-tune the machine-translated text” ( Puppel & Borg, 2024 , p. 22). Applying his study on three machine-translated short stories from English into Dutch, Macken (2024) evaluated the ability of ChatGPT 4-o, in a post-editing machine, to translate literary texts automatically in comparison to the post-edition of experienced professional translators in the field of translating literary works from English into Dutch. The study showed that the automatic changes made by ChatGPT were at the level of words and that it made more lexical changes than those made by human editors. In contrast, professional editors made changes not only at the word level but also at the style level. Overall, the study concluded that although ChatGPT could actually correct a number of errors, it still provided edited texts with more problems than texts post-edited by professional translators. In a comparative study of the advantages and limitations of conventional MT systems, new chatbots, and ChatGPT tool in the translation from English into Spanish and Portuguese, Postigo (2024) stated that there are differences between the translation output obtained by MT and those gained by AI tools. Nevertheless, she found that all MT and AI tools present several grammatical and semantic challenges. Additionally, Jiang et al. (2024) conducted a quantitative, qualitative study on the potentiality of the automated evaluation of machine translation from Chinese into Portuguese by means of two ChatGPT models, i.e., 3.5 and 4.0, in addition to five human raters, for the sake of analytic comprehensiveness. The sample consisted of 20 sentences translated from Chinese into Portuguese. The study concluded that the capability of ChatGPT, especially the 4.0 model, reached the efficiency level of conventional human evaluation and that it has a more inspiring future if it gets more enriched with further balanced capabilities. To detect the cultural integration of AI tools in their translations, Cao et al. (2024) addressed the adaptation of nuanced cultural contexts by translating Chinese recipes to English. These recipes are both automatically formed and human-enriched. The study concluded that, although GPT-4 showed mesmerizing capability in the adoption of cultural recipes, it still remained behind the scale of human ability and expertise. Additionally, Deilen et al. (2023) conducted an intralingual German-Easy Language machine translation and translated it using ChatGPT. The analysis was based on readability, correctness, and syntactic complexity. The results showed that the output content was not completely correctly intra-lingually translated, and some content was missing, while syntactic constructions were rendered easier but not as required. The paper concluded that there is an inevitable need for professional human translators to pre-instruct the tool and edit the tool’s translated output. 3.1 Translating Arabic dialects Early work on translating Arabic dialects into foreign languages has predominantly relied on Standard Arabic (SA) as a pivot language and on hybrid or statistical machine translation approaches. For example, Sawaf (2010) proposed a hybrid MT system combining rule-based and statistical methods, showing that normalizing dialectal input into MSA significantly improves translation quality across multiple Arabic dialect groups. Similar normalization-based strategies were adopted to reduce out-of-vocabulary (OOV) rates and enhance BLEU scores, particularly for noisy web and broadcast data ( Salloum & Habash, 2012 ). These studies collectively demonstrate that character-level transformations, morphological analysis, and dialect-to-MSA mapping are effective for mitigating lexical and orthographic variation in dialectal Arabic. Subsequent research moved beyond single-system approaches by explicitly addressing dialectal variation and system complementarity. Salloum and Habash (2012) showed that generating multiple MSA paraphrases for dialectal input and using them in SMT improves translation quality, while Salloum and Habash (2012) introduced sentence-level dialect identification to dynamically select the most appropriate MT system. Large-scale efforts under DARPA’s BOLT program further confirmed that combining MSA and dialectal data, applying morphological segmentation, and carefully balancing training corpora yield superior performance for informal Arabic texts that mix registers and dialects ( Zbib et al., 2012 ; Aransa, 2015 ). These findings underscore the importance of treating Arabic dialects and SA as related but distinct linguistic domains in MT. More recent studies have focused on morphological segmentation, OOV handling, domain adaptation, and non-standard writing systems. Unsupervised segmentation techniques were shown to improve translation quality for dialect-to-English and English-to-dialect MT ( Al-Mannai et al., 2014 ), while targeted OOV normalization using dialect identification and morphological tools yielded consistent gains ( Durrani et al., 2014 ). In addition, research has highlighted the growing challenge posed by Arabizi, a non-standard Latin-script representation of Arabic dialects widely used in social media, which introduces substantial orthographic variability and further complicates MT for dialectal Arabic. Thus, the literature converges on the view that effective Arabic dialect MT requires dialect-aware preprocessing, adaptive modeling strategies, and robust handling of linguistic and orthographic variation. In his study, Alafnan (2024) detected the effectiveness of ChatGPT and Google Translate in the translation of selected Arabic and English speeches of the King of Jordan, Abdullah II, from Arabic to English, and from English to Arabic. This study revealed that Google Translate’s translated outputs were inadequate and required major editing. Translated outputs of ChatGPT, specially from Arabic into English, on the other hand, though needing some sort of adjustment, were acceptable. However, the study emphasized that machine translation complements human professional translators and does not substitute them, since human mediation is indispensable for the adequacy of translation. Thus, the study has three questions to answer: 1. Can AI models, specifically ChatGPT and DeepSeek, capture the YSA dialectical nuances? 2. Which is better in translating YSA terms, ChatGPT or DeepSeek? 3. What are the problematic nuances that ChatGPT and DeepSeek face in translating such terms? 4. Methods 4.1 Data collection We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of San’ani Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow for both linguistic and cultural analyses. After collecting the data, the terms were classified into categories. The results are presented in Table 1 . Table 1. Categories Yemeni San’ani Arabic terms. Nominals Nonnominals Places Clothes Animals Stuff Households Verbs Adj & Advs شاقوص مُغْمُق دِمِّة شِرْكِة طاسة نِبْدَع مطنن صومعة مَحْزَق حلباني جَعالة بالدي وخّر شوعة دَيْمِة صُماطة تيس قِرِّيْح بُورِي قوٌّى حالي قاع أدوان معزة تـُتنْ مَعْشَرَة كوزر بحين غُرقة مشمع شقري بيعة مَدَق سير أيحين دِرجان بَرْدِة مَلَتْ بيشقص فيسع حانوت نصلة قوطي يلغّج لَلْمه زنة المدعي سمخ سَعْما محواش As shown in Table 1 , the data were divided into seven types: Place, Clothes, Animals, Stuff, Household, Verbs, and Adj & Advs. The idea behind this categorization is twofold: (i) easy to expose and (ii) to understand the nature of each category. The category Place contained seven terms: Clothes 8, Animals 5 Stuff 5, Households 9, Verbs 8, and Adjs & Advs 8 terms. 4.2 Procedure The study has passed through three stages. Stage 1 concerns data collection and classification, in which we collected the data from different sources, as alluded to above, and classified these data. Stage 2 consists in translating the collected data into English using ChatGPT-4o and DeepSeek v3. After translating the data using both AI models, we translated the terms. Our translation depended heavily on two factors: i) our knowledge of YSA and ii) the first author’s relatives who are native speakers of YSA. If we did not know what a term meant, we asked them to give us the meaning in SA or explain its meaning by giving us an example, for instance. When we understand the meaning, the translation task becomes easier. We then tabulated the results in terms of ChatGPT, DeepSeek, and human translations. Stage 3 deals with the data analysis and discussion. 4.3 Methods of analysis The methodology employed was a qualitative and quantitative approach to evaluate the accuracy and contextual appropriateness of translations. In our analysis, we followed a simple count and percentage for quantitative analysis. For the qualitative part, we adopted a linguistic and sociocultural analysis framework, known as micro and macro analysis (see e.g. Fairclough, 2003 ). We assessed the translation (in)correctness and appropriateness of dialectical terms between the two AI models, ChatGPT and DeepSeek, and human translation. We considered the translation of an item correct if cultural and linguistic aspects are both maintained in the translation of this term. However, if one of these aspects is violated in the translation, we consider it appropriate, and if both the cultural and linguistic aspects are violated in the translation, we considered it incorrect. We set the following evaluation criteria: i) dialectal and cultural subtleties, in which we examined whether both AI models capture these nuances and which are more successful, and ii) linguistic subtleties. Using this criterion, we assessed whether both models capture linguistic subtleties that lie within morphophonology/orthography and syntax) contextual subtleties. Here, we investigated whether the models’ translations reflect the intended meaning in different contexts, and iv) Standard Arabic (SA) biases, where we evaluated whether the translations produced by both models are biased towards SA. 5. Results and discussion 5.1 Results We first present the results in the form of ChatGPT, DeepSeek, and human translations. Table 2 presents these findings. Table 2. AI Translation (ChatGPT and DeepSeek) and Human translation. No Term ChatGPT DeepSeek Human 1 شاقوص Hook (or a type of tool with a hooked end) A type of fish (likely a local or dialectal term) A small and narrow window 2 صومعة Silo/Minaret/Monastery Silo or minaret minaret 3 دَيْمِة Continuous rain (light drizzle) Light rain or drizzle Kitchen 4 قاع Valley/Lowland/Open land Bottom or base Bottom 5 غُرقة Room Drowning or flood hole 6 دِرجان Stairs Stairs or steps Stairs 7 حانوت Shop/Store Shop or small store shop 8 مُغْمُق Dark/Dim Confused or muddled Veil 9 مَحْزَق Tight place/Constricted area Narrow or tight spot belt 10 صُماطة Leftover food/Small food scraps A small bag or pouch shawl 11 أدوان Enemies Tools or utensils clothes 12 مشمع Waxed cloth/Plastic cover Oiled cloth or tarpaulin Plastic cover 13 بَرْدِة Curtain A type of cloak or garment Curtain 14 نصلة Blade/Tip of a knife or sword Arrowhead or sharp point Djambia 15 زنة Weight Weight or measure garment 16 دِمِّة Silence/Stillness A type of plant or herb (context-dependent) cat 17 حلباني Milkman Related to milk or milking Millipede 18 تيس Male goat Male goat Male goat 19 معزة Female goat Female goat Female goat 20 شقري Blond/Fair-skinned Chestnut-colored (often used for horses) Cock 21 شِرْكِة Company/Partnership Partnership or company meat 22 جَعالة A small reward or treat Reward or compensation Sweet presents 23 قِرِّيْح Wound/Injury A type of plant or herb firecrackers 24 تـُتنْ Tobacco Tobacco Tobacco 25 بيعة Sale/Transaction Oath of allegiance or sale leek 26 نِبْدَع We start/We begin We innovate or create We start 27 وخّر Move away/Step aside He delayed or postponed Step aside 28 قوٌّى Strengthened/Madestronger He strengthened or empowered welcome/please 29 كوزر Drank excessively A type of plant or herb sit 30 سير Go/Walk/Travel Walk or journey go 31 بيشقص Cuts into pieces/Trims He cuts or slices oversee 32 يلغج Chews noisily/Mumbles while eating He combs or untangles Talkative 33 طاسة Bowl/Metal cup A small bowl or cup bowl 34 بالدي Cold/Chilled Cold or chilly Bucket 35 بُورِي Pipe/Exhaust pipe A type of mat or straw Tobacco container 36 مَعْشَرَة Gathering/Assembly Companionship or group Big and wide bowl 37 مَدَق Mortar and pestle (used for grinding) He pounded or crushed pestle 38 مَلَتْ Got bored/Fed up She tilted or inclined Wooden pot used to mix tobacco 39 قوطي Can/Tin A type of plant or herb Can 40 مدعي Pretender/Faker Claimant or plaintiff hookah 41 محواش Farmyard/Enclosure He erased or removed Small wooden mixer used for Aseed 42 مطنن Distracted/Not paying attention He exaggerated or embellished upset 43 شوعة Bright light/Flash A small branch or twig ugly 44 حالي Sweet/Delicious My condition or state beautiful 45 بحين When/At the time At the time or meanwhile fast 46 أيحين Right now/At this moment When or at what time when 47 فيسع Quickly/Fast He expands or makes room quickly 48 لَلْمه Gather it up/Collect it A type of plant or herb why 49 سَعْما Sometimes A type of plant or herb like 50 سمخ Jumped/Leaped He raised or elevated brave In Table 2 , there are several types of translations resulting from the translation of both ChatGPT and DeepSeek. Both models provided correct translations, incorrect translations, and appropriate translations. For some terms, they provide both correct and incorrect translations simultaneously. As for ChatGPT translations, we have correct translations including translating term 1, viz. شاقوص which was translated by ChatGPT as ‘Hook (or a type of tool with a hooked end) ’. The incorrectness of translation here lies in rendering the Yemeni San’ani Arabic term with Hook which is far away from the correct translation. However, humans render it a small and narrow window which is correct. The YSA term شاقوص is found in old houses, specifically those in the Old Sana’a City. It is located on the stairs (or even in rooms) and is used by a person to see through, but without being noticed by another person. For example, if someone knocks on the door, and only women in the house, they used شاقوص to see who is the knocker before opening the door. To exemplify the correct translations by ChatGPT, take the term دِرجان which means in SA دِرج , a plural form of دِرجة ‘stair.’ As for appropriate translations of ChatGPT, take, for example, the term جَعالة which is a mixed sweet present including biscuit, sweets, brought by a father, mother, or older brother for children. ChatGPT translates this term as a small reward or treat , which is acceptable but not absolutely correct. This is rendered through human translation. Additionally, DeepSeek translations result in several types of renderings. We obtained seven correct translations, 34 incorrect translations, and eight appropriate translations. Like ChatGPT, DeepSeek provides both correct and incorrect translations simultaneously. DeepSeek has several correct translations. For example, the term أيحين was translated as When or at what time. أيحين is in fact a wh-word in YSA which is used for asking about time (see also Shormani et al., 2025 ). To exemplify the incorrect translations by DeepSeek, the term شاقوص was translated as ‘a type of fish.’ This translation is far from the correct translation, which is ‘a small and narrow window.’ As for appropriate translations by DeepSeek take, for instance, the term بَرْدِة which means ‘curtain’ as in the human translation, but DeepSeek translates it as ‘a type of cloak or garment.’ This translation is acceptable due to the fact that بَرْدِة is a piece of cloak or garment used to cover windows. DeepSeek also provides both correct and incorrect translations as in the case of صومعة which was translated as Silo and minaret , the first of which is incorrect while the latter is correct. That the rendered term ‘Silo’ is incorrect is due to the fact that ‘Silo’ is used for keeping corps after harvesting, while ‘minaret’ (of a mosque) is what is meant here. Table 3 displays the frequency and percentage of correct, incorrect, and appropriate translations of both ChatGPT and DeepSeek. It also presents the frequency and percentage where both models provide correct and incorrect translations simultaneously. For ChatGPT, there were 17 (34%) correct translations. It translated 29 terms incorrectly and 58% of the total number of terms involved. There are three appropriate translations by ChatGPT. The term for which ChatGPT provides correct and incorrect translations at the same time is only 1, namely صومعة as has been discussed above ( Table 3 ). DeepSeek differs from ChatGPT. For example, it has only seven terms that were translated correctly, that is, less than ChatGPT, amounting to 14% of the total number of terms involved in the study. There are 34 terms that were incorrectly translated by DeepSeek (68%). Appropriate translations scored eight terms, that is, 16%. Finally, DeepSeek provides both correct and incorrect translations for term 1, namely صومعة as just noted with regard to ChatGPT. Interestingly, both AI models provided correct and incorrect translations for the same term. Table 3. Summary of the results. ChatGPT DeepSeek Tra. category Freq % Freq % Correct 17 34 7 14 Incorrect 29 58 34 68 Appropriate 3 6 8 16 Co&inc 1 2 1 2 5.2 Discussion 5.2.1 ChatGPT translations As shown in Table 2 , ChatGPT yielded 18 correct translations. These correct translations include دِرجان , translated as ‘stairs.’ This is accurate because in Standard Arabic, دِرجان is the plural form of دِرجة , meaning ‘stair.’ This term is widely used in YSA to refer to a set of stairs leading to another floor of a multifloored building. Another accurate translation is that of قاع , rendered as ‘valley/lowland/open land’. In YSA, قاع refers to a flat, low-lying area of land, often used for agriculture. The term is commonly used in regions where such landscapes are found, where ChatGPT translation is correct. Similarly, حانوت was correctly translated as ‘shop/store’. This word, originating from “older” Yemeni Arabic, remains in use in YSA to describe a small store or shop selling various goods. This aligns well with ChatGPT translation. Another well-rendered term is تـُتنْ , translated by ChatGPT as ‘tobacco.’ In YSA, تـُتنْ is a commonly used word in YSA. تـُتنْ is used as the substance for smoking hookah, making this translation highly accurate. The term بَرْدِة was also correctly translated by ChatGPT as ‘curtain.’ The term is widely used in YSA; it is part of households referring to a piece of cloth covering a window or doorway. Additionally, the term نِبْدَع was accurately translated by ChatGPT as ‘we start/we begin’. In YSA, it is used to indicate the initiation of an action or event, such as starting work or journey. Another correct translation is سير , translated as ‘go/walk/travel’ though ‘go’ is the best translation, and this is what San’anis mean when using it. The term ‘walk’ has another term in YSA, which is إخطى . In YSA, سير is a commonly used verb to tell somebody ‘to go,’ specifically by walking. Also, طاسة was correctly translated by ChatGPT as ‘bowl/metal cup’. In YSA, it refers to a small, often metallic container used for drinking (soup) or eating. The term بُورِي was incorrectly rendered as ‘pipe/exhaust pipe’. The term بُورِي is a word for the container of تـُتنْ which is put on hookah. Likewise, مَدَق was correctly translated as ‘mortar and pestle’ (used for grinding). It is often made of copper or metal, an essential kitchen tool in Yemen used for grinding spices or grains. The term قوطي was also correctly translated as ‘can/tin’. We now turn to ChatGPT incorrect translations. This category included 29 terms. For example, ChatGPT translated شاقوص as ‘Hook (or a type of tool with a hooked end),’ which is incorrect. In YSA, شاقوص refers to a small window found in traditional houses, particularly in the Old Sana’a City. These windows are strategically placed on stairs or rooms, allowing residents to see outside without being noticed. For example, if someone knocks on the door and only women are home, they use the شاقوص to identify the visitor before opening the door, as noted to above. This highlights the cultural and architectural significance of the term that ChatGPT missed. ChatGPT translated دَيْمِة as ‘Continuous rain (light drizzle),’ which is incorrect. The human translation, ‘kitchen,’ is accurate. Though the term مطبخ , is used in other parts of Yemen, there are also old Yemeni terms used for ‘kitchen’ in some parts of Yemen, as in the case of سقيفة which is used in Ibb region., For instance, the term دَيْمِة is widely used in YSA to mean ‘kitchen’. This term is deeply tied to daily life in YSA, and this ChatGPT translation is unrelated to its actual meaning. ChatGPT translated غُرقة as ‘room,’ which is incorrect. The human translation, ‘hole,’ is accurate due to the fact that غُرقة refers to a hole in YSA. ChatGPT translation of this term reflects a lack of understanding of the term’s true meaning. It translated مُغْمُق as ‘dark/dim,’ which is incorrect, compared to the human translation, ‘veil,’ which is what this term in YSA refers to. مُغْمُق refers to a woman’s veil, which San’ani women used to cover their faces. It is a piece of cloth to covering women’s faces, having cultural and religious significance in Yemeni society. It seems that ChatGPT translation of this term does not capture the term’s cultural and religious connotations. ChatGPT translated مَحْزَق as ‘tight place/constricted area,’ which is incorrect. The incorrectness of this translation lies in capturing neither the cultural nor the dialectical nuances. In YSA, the term مَحْزَق refers to a belt decorated with gun bullets’ holes. In Sana’a, and northern places in Yemen, مَحْزَق is used as a sign of “manhood,” courage and a signal of fighting. مَحْزَق is often worn with a gun. All these features of مَحْزَق are reflected in the human translation. ChatGPT translated شِرْكِة as ‘company/partnership,’ which is far from the dentation of this term in YSA. The term شِرْكِة simply means ‘meat’. refers to meat, a staple in yemeni cuisine, and daily life. ChatGPT translation reflects a misunderstanding of the term, likely due to the word’s similarity to the Arabic term for ‘partnership’ ( شركة ), but with orthographical differences. The latter is written as شَرٍكَة , note the different ‘harakat’, as we will see shortly. Finally, the appropriate (acceptable) translations by ChatGPT included only three terms. In this category, ChatGPT provides a translation that is somewhat acceptable but not fully accurate as in translating the term جَعالة , translated by ChatGPT as ‘a small reward or treat’. While this conveys a general idea, the specific meaning in YSA culture refers to a mix of sweets, biscuits, and treats brought by a father, mother, or older sibling for children as a gesture of care and love. It is a traditional and cultural practice rather than just a generic ‘treat.’ Another example is بحين , translated as ‘when/at the time’. While this is an acceptable translation, the term in YSA more precisely means ‘as soon as’ or even ‘fast’. Similarly, أيحين was translated as ‘right now/at this moment’, which is mostly correct. However, in YSA, أيحين carries a sense of questioning, meaning ‘when’ as can be observed in human translation. 5.2.2 DeepSeek translations Recall that DeepSeek has seven correct translations, 34 incorrect and eight appropriate translations, and both correct and incorrect translations. We exemplify some of these correct translations. For example, DeepSeek translated the term قاع , as ‘bottom or base.’ This is accurate because in YSA, قاع also refers to the lowest part of something such as the bottom of a container. However, as we have seen in ChatGPT translation, the term قاع refers also to a valley which is fertile for agriculture. There are also many well-known قيعان ‘plural of قاع ” in Yemen such as قاع البون، قاع جهران (Jahran valley, Bawn valley, respectively), etc. There are several well-known fertile valleys in Yemen. Almost all types of crops are grown in these fertile valleys. Another term translated correctly by DeepSeek is دِرجان ‘stairs.’ DeepSeek adds ‘steps’ as an alternative translation which does not reflect the actual context like ‘stairs.’ حانوت was also accurately translated as ‘shop or small store,’ reflecting its meaning in the local dialect. Note that this translation aligns with ChatGPT translation. DeepSeek adds ‘small’ which is true; in San’ani dialect, حانوت is a small shop in a traditional old market like Souq almilħ, a famous Souq in Sana’a Old City. Additionally, تيس was correctly rendered as ‘male goat,’ and معزة as ‘female goat,’ both of which match the human translations. Another accurate translation is تُتن , which DeepSeek translated as ‘tobacco,’ a common term in YSA for tobacco products. However, several incorrect translations were made. The term شاقوص , for example, which DeepSeek translated as ‘a type of fish (likely a local or dialectal term)’ is a good case in point here. This is incorrect, as شاقوص in YSA dialect refers to a small and narrow window in old houses, particularly in Old Sana’a City, as we have noted earlier in relation to ChatGPT translation. The term صومعة in addition was mistranslated as ‘light rain or drizzle,’ while in reality, it means ‘minaret,’ i.e. a tall circled narrow tower of a mosque. Another significant mistranslation is دَيْمِة , which was rendered as ‘confused or muddled,’ whereas in YSA, it actually refers to kitchen. Unlike ChatGPT, DeepSeek translated the term صُماطة as ‘a small bag or pouch,’ which is not correct. It refers to some sort of ‘shawl,’ a traditional Yemeni piece of clothes that covers men’s heads, or put on their shoulders. Another incorrect translation concerns the term زنة , which DeepSeek rendered as ‘weight or measure,’ while its actual meaning in YSA is ‘garment’ like robe. دِمِّة was mistranslated as by DeepSeek as ‘a type of plant or herb,’ but it actually means ‘cat’ as in the human translation of the term. The term بُورِي was translated as ‘a type of mat or straw,’ which is not correct. The correct translation is the container of تـُتنْ which is put on hookah, as we have noted so far. DeepSeek also mistranslated يلغج as ‘he combs or untangles,’ whereas it actually means to repeat telling something several times or simply ‘talkative’ in YSA. بالدي was translated as ‘cold or chilly,’ but it means ‘bucket’ in San’ani dialect. مَعْشَرَة was mistranslated as ‘companionship or group,’ but its actual meaning is ‘big and wide bowl,’ which is used for serving Aseed in San’ani dialect. مَدَق was translated as ‘he pounded or crushed,’ but it refers to a small wooden or copper mixer used for grinding especially spices as indicated in the human translation. مَلَتْ was rendered as ‘she tilted or inclined,’ whereas its correct meaning is a wooden or aluminum container in which tobacco is mixed. Regarding appropriate translations, DeepSeek translated غُرقة as ‘drowning or flood,’ which is somewhat close but not fully accurate, as its proper meaning is ‘hole. أدوان was translated as ‘tools or utensils,’ which is acceptable but not entirely a precise translation, as its actual meaning is clothes. مشمع was translated as ‘oiled cloth or tarpaulin’, which is reasonable but slightly different from the correct meaning of ‘plastic cover. بَرْدِة was translated as’ a type of cloak or garment’, while it actually means ‘curtain’. نصلة was translated as’arrowhead or sharp point’, which is somewhat close but does not fully convey its actual meaning, Djambia (the traditional Yemeni dagger). جَعالة was translated as’reward or compensation’, but it should be understood as a mixture of sweets, biscuits, and treats given to children. سير was translated as ‘walk or journey,’ which is close but does not capture its everyday use in YSA as ‘go.’ بحين was acceptably translated as ‘at the time or meanwhile,’ though the human translation conveys the meaning more naturally. 6. ChatGPT and DeepSeek: Convergence and divergence 6.1 Dialectal and cultural subtleties Language reflects culture and vice versa (see e.g. Newmark, 1988 ; Bassnett, 2013 ; Shormani, 2020 ), and given that dialect is a regional variety of a language, one could argue that dialect could reflect culture more precisely than language because it represents a regional identity ( Arzu & Issa, 2014 ; Daulay, 2017 ). A dialect carries regional variations mirroring everyday affairs, people’s needs, and emotions. Thus, the dialect is closer to people than to language. In our case, YSA is closer than SA to Sana’ anis, as it is their mother tongue. A dialect may also include lexes that are not found in a standard language. For example, the term جَعالة belongs from YSA, but not SA. Given these dialectical peculiarities, we examined how ChatGPT and DeepSeek deal with such dialectical subtleties and whether they both understand them. For instance, the term جَعالة refers to a gift of sweets for children brought by father, mother, or relatives. ChatGPT translates it as ‘a small reward or treat,’ which is close to the meaning, whereas DeepSeek mistranslates it as ‘reward or compensation,’ losing the cultural sense. The term نِبْدَع was accurately translated by ChatGPT as ‘we start/we begin’. However, DeepSeek was not able to translate and capture this dialectical aspect, hence providing an incorrect translation, ‘we innovate or create.’ Both ChatGPT and DeepSeek incorrectly translated حلباني as ‘milkman,’ and ‘elated to milk or milking,’ respectively, which shows their failure to capture the cultural and dialectical nuances of this term. The term حلباني in YSA refers to an insect commonly found in Yemen, specifically in rain seasons, whose English equivalent is ‘millipede. Both ChatGPT and DeepSeek also translated شقري as ‘blond/fair-skinned’ and ‘chestnut-colored’. Neither term reflects the correct dialectical or the Sana’ni Yemeni culture. شقري refers to a living cock, specifically when sold/bought in the market. Another term that has been mistranslated by both models is مدعي . ChatGPT translated it as ‘retender/faker’ while DeepSeek ‘claimant or plaintiff.’ Both translations are neither correct nor appropriate, as can be observed from contrasting them with the human translation, viz., hookah. Another example is محواش which manifests YSA both dialect and culture. As for dialect, the term محواش refers to a wooden tool used for mixing Aseed (see here ). However, the same tool is referred to as مجحي ‘majhi’ in Ibbi dialect, for instance. Here lies the concept of dialectical difference. Additionally, this word may not belong to SA because, as far as we can tell, there is nothing called “Aseed” in SA. “Aseed” is a Yemeni dish; no other Arab country has it, and here lies the cultural peculiarities of the term “Aseed,” in general, and محواش , in particular. A further term reflecting San’ani dialect and culture is لَلْمه , which is a San’ani Arabic-specific term, and San’anis are sometimes mocked by other Yemenis belonging to other Yemeni regions. This term is an adverb, specifically a wh-word meaning ‘why.’ A final term that reflects both San’ani dialect and culture that can be highlighted here is مغمق . As we have discussed so far, the term مغمق simply means ‘veil,’ though the “veil” is used in all Yemen, مغمق has a specific (dialectical) cultural connotation. In Sana’a, ‘veil’ is different from any ‘veil’ used in other Yemeni regions; it is piece of cloth worn by women only in Sana’a governorate (and some parts of Amran, a governorate which was part of Sana’a (which has recently become an independent governorate). 6.2 Linguistic subtleties In this category, we examine the ability of both models to capture linguistic nuances in terms of syntax and morphophonology/orthography. Regarding the former, there are four syntactic categories in our data: nouns, verbs, adjectives, and adverbs. The category nouns includes most of the terms involved, such as Place, Clothes, Animals, Stuff, and Households. The latter is discussed in terms of the morphophonology/orthography features that these terms involve. 6.2.1 Syntax In our corpus, the syntactic category, which includes eight verbal terms, seems to be the most difficult syntactic category for both models. For example, while ChatGPT has 3 verbal terms (out of 7), viz., وخّر , نِبْدَع , and سير translated correctly, DeepSeek has no correctly translated verbs. It has only one verb which was translated appropriately, namely سير . Additionally, the most translated category by both models seems to be nouns, although ChatGPT scores more correct translations than DeepSeek. While the former translated 15 nouns and one adjective and one adverb correctly (as marked in black), the latter translated six nouns and only one adverb correctly. 6.2.2 Morphophonology/orthography To correctly provide the dialectical representation of some words in our data, we provided diacritics, known in Arabic as harakat to differentiate them from their SA equivalents or to mark the Sana’ ani-specific peculiarities. These include َ , ِ , ُ , and ْ ( fataha, kasrah, dhumah, and sukun , respectively) (see e.g. Shormani, 2013 ). These harakats, except sukun can be equalized to the English short vowels a, i, u, respectively. They are placed on letters to indicate how they should be pronounced. Words having these harakats include تـُتنْ , شِرْكِة , بَرْدِة , مُغْمُق and مَعْشَرَة . For example, in SA, we have the term نُبْدِع which means’we innovate’, which in turn is different from the YSA نِبْدَع . Linguistically, the two words have different pronunciations. Most of these words were mistranslated by both models ( Table 2 ). However, ChatGPT seems to capture the morphophonology/orthography features of these words more than DeepSeek. 6.3 Contextual subtleties The SL context in which a word is used is considered a crucial issue in translation, as it transfers the intended meaning to the TL audience. The fact that a specific term is used in more than one context is conveyed by the operator “or” or “/”. ChatGPT used “or”/ “/” 39 times, but DeepSeek used “or” 47 times ( Table 2 ). Considering this aspect, in addition to the (in) correctness of translations provided by both models, there seem to be two contradictory aspects: i) regardless of the (in) correctness of translations, it seems that both models are “aware” of the context of terms, providing more than one translation for a considerable number of terms, that is, 39 vs. 47. In this very aspect, DeepSeek seems to capture context better than ChatGPT does. And ii) if, however, we consider the (in) correctness of the translations, it seems that ChatGPT captures dialectical and cultural nuances more than DeepSeek. The number of correct (and incorrect) translations by ChatGPT and DeepSeek gives us a clear clue that ChatGPT demonstrates a stronger grasp of the cultural and dialectical contexts of YSA than DeepSeek, as it has 17 correct translations, while DeepSeek has only seven correct translations. 6.4 Standard Arabic biases Given that both models’ incorrect translations are more than their correct ones, although ChatGPT succeeds in capturing YSA nuances to some extent, it seems that both models are SA-biased. Both models often seem to default on SA or broad literal meanings (cf. Alwagieh & Shormani, 2024 ). For instance, both models translate صومعة incorrectly. While DeepSeek translates it as ‘light rain or drizzle,’ ChatGPT translates it as ‘continuous rain,’ both missing the true meaning ‘minaret.’ The term شقري , which means ‘cock’ in YSA, is mistranslated by DeepSeek as ‘chestnut-colored,’ a more generic SA meaning, perhaps from the SA term أشقر ” blond’. ChatGPT makes a similar mistake with ‘blond/fair-skinned’. It seems that both models are SA-biased; both retain a link to color-based descriptions of the SA meaning, showing a clear SA bias. Another term to be considered here showing SA bias is قوّى translated as ‘strengthened/made stronger’ and ‘he strengthened or empowered’ by ChatGPT and DeepSeek, respectively. Apart from its correct dialectical meaning ‘welcome,’ both models’ translations, though somehow different, seem to take SA meaning considerably in translating this term. Note that the term قوّى can also mean ‘please’ in YSA, as reflected in the human translation. The SA term قوي ‘strong’ seems to influence both models’ translations, again demarcating their SA bias. Thus, this SA bias could be ascribed to the training data, that is, both models appear to be trained only on SA data. 7. Conclusions, recommendations and limitations The findings demonstrate a clear disparity in the models’ abilities, with ChatGPT consistently outperforming DeepSeek in capturing the YSA-specific dialectical features. Several conclusions could be drawn from this study. First, dialectal and cultural subtleties pose considerable challenges. Both models struggle significantly with dialectal and culturally embedded terms, although ChatGPT demonstrates a better approximation of meaning in several instances. This is most evident in culturally loaded terms such as نِبْدَع , جَعالة , and محواش , where ChatGPT translations are closer to the intended meanings, while DeepSeek often fails to capture cultural connotations. However, in several cases such as مدعي , شقري , and مغمق , both model fail to provide culturally and contextually appropriate translations, highlighting a systemic challenge in handling regionally specific lexicon. Difficulty is particularly centered around the “stuff” category, which includes highly dialectical and culturally specific items. Culture-based terms are reported as difficult to translate, even for advanced English students (see Alshawsh & Shormani, 2025 ). None of the five terms in this category were correctly translated by either model, emphasizing a significant gap in the ability of current AI models to handle deeply localized cultural elements (cf. Lilli, 2023 ; Datta, 2023 ). Second, it is clear that ChatGPT outperforms DeepSeek in linguistic subtleties. For example, in terms of syntactic categorization, both models struggled most with verbs, which are often morphologically complex and context dependent. While ChatGPT translated three out of the seven verbs correctly, DeepSeek succeeded in only one instance, showing a stark contrast in performance. Third, ChatGPT outperforms DeepSeek concerning the morphophonological and orthographic distinctions. ChatGPT exhibits high performance over DeepSeek. The use of diacritics (harakat) in the dataset helped distinguish YSA terms from their SA counterparts (e.g., نِبْدَع vs. نُبْدِع ), yet most of these marked terms were still mistranslated by both models. However, ChatGPT handles these distinctions with relatively higher accuracy, suggesting a more nuanced internalization of orthographic and morphophonological cues. Fourth, while both models exhibit some degree of contextual awareness, as seen in their use of multiple translation options (i.e., use of “or” & “/”, ChatGPT: 39 instances, DeepSeek: 47), this does not necessarily translate into correct translations. DeepSeek’s higher frequency of “or” usage suggests greater surface-level awareness of polysemy or ambiguity, but this does not correlate with translation accuracy. The final conclusion concerns the SA bias exhibited by both models. This is particularly visible in the mistranslations of terms such as قوى , شقري , and صومعة , where both models defaulted to SA meanings that do not reflect the YSA usage. This SA bias can be attributed to the composition of the model training data, which are likely dominated by SA sources. Consequently, dialectical terms that deviate from SA norms are either misinterpreted or forcibly aligned with their SA counterparts, leading to semantic distortions and a lack of cultural and contextual fidelity. Thus, these aspects require careful attention from the AI developers. AI developers should expand the dialectal data coverage in the training data to improve the performance of both AI models in translating dialects. For example, in the San’ani dialect, data should be collected from various sources such as recorded conversations, social media content, and dialect-specific literature or oral history (see e.g. Morano et al., 2025 ). AI developers are advised to incorporate cultural and social contexts into training data. Many dialectal expressions are deeply rooted in the local culture and social norms. Therefore, AI models should be designed to consider these contexts to avoid literal or inaccurate translations that miss the implied meanings or connotations. Additionally, AI models, here ChatGPT and DeepSeek, should include or improve features that can automatically detect the dialect used in the input text. This would enable a more precise adaptation of translation strategies specific to the YSA dialect (or other dialects across several languages), improving the overall output quality. AI developers should consider enabling community- or researcher-led fine-tuning of models in niche dialects. They could also encourage (and perhaps fund) studies that tackle dialectical varieties (see also Kadaoui et al., 2023 ). Thus, we propose ‘a “multi-dialectal” pre-training approach’ (see e.g. Zan et al., 2022 ) incorporating YSA data as well as other Yemeni dialects. And this “multi-dialectal pre-training” could be extended to all Arabic multi-dialects such as Egyptian Arabic, Gulf Arabic, Moroccan Arabic. Another well-documented method that could be proposed here is a Bidirectional Training (BiT) approach (see e.g. Ding et al., 2021 ). This approach could be fine-tuned utilizing ‘both YSA-to-English and English-to-YSA data’ to enhance LLMs’ ability to translate YSA terminology into English, grasping YSA’s linguistic and cultural nuances. However, this study has some limitations. The first limitation is that it focuses exclusively on the San’ani dialect. While this dialect is widely spoken, the results may not generalize to other Yemeni dialects, such as Ibbi, Adeni, Hadhrami, or Arabic dialects from other regions, which may exhibit distinct lexical, morphophonological, or syntactic features. The second limitation concerns the sample of the dialectal terms and phrases involved. It was relatively limited in size. A larger and more diverse dataset can improve the reliability of the findings and better reflect the full linguistic complexity of the dialect. Third, the study involved only two AI translation models: ChatGPT and DeepSeek. While these models are prominent and state-of-the-art, other platforms, such as Grok 3, Felo, and Meta, could also be used in future studies to widen the breadth of comparative analysis. Fourth, we used isolated words only, which may affect the models’ ability to contextualize the meaning. Specifically, LLMs are reported to be more accurate on contextualization than in single words. Thus, the two models’ performance may change if we use phrases and sentences. Fifth, our prompt in asking ChatGPT and DeepSeek to translate the 50 terms could have been more explicit if we give some explanations of what we need or employing COMET (see e.g. Peng et al., 2023 ). Finally, AI models such as ChatGPT and DeepSeek are frequently updated; hence, the performance observed during this study may not remain static over time, potentially affecting the reproducibility or relevance of the results for future users. Ethics and consent No ethics and consent statements are required for this study. Data availability statement The data underlying the results of this study are available on figshare.com , entitled Translating dialects DOI: https://doi.org/10.6084/m9.figshare.29251088 ( Shormani & Al-Samki, 2025 ). Data are available under the terms of the Creative Commons Attribution 4.0 International license (CC-BY 4.0). References Alafnan M: Large Language Models as Computational Linguistics Tools: A Comparative Analysis of ChatGPT and Google Machine Translations. J. Artif. Intell. Technol. 2024; 5 : 20–32. Publisher Full Text Ali G, Ali N, Syed K: Understanding Shifting Paradigms of Translation Studies in 21 st Century.2023. Publisher Full Text Al-Mannai K, Sajjad H, Khader A, et al. : Unsupervised word segmentation improves dialectal Arabic to English machine translation. Proceedings of the EMNLP 2014 Workshop on Arabic Natural Language Processing (ANLP). 2014; pp. 207–216. de Almeida G , O’Brien S: Analysing post-editing performance: correlations with years of translation experience. Proceedings of the 14th Annual Conference of the European Association for Machine Translation, St. Raphaël, France. 2010; pp. 27–28. Alshawsh H, Shormani MQ: (Un) translatability of Yemeni (Ibbi) Zawaamil and Ballads into English: Ibb University Students as a Case Study. Int. J. Linguist. Lit. Transl. 2025; 8 (3): 188–205. Publisher Full Text Alwagieh N, Shormani MQ: Translating Arabic free poetry texts into English by ChatGPT: Success and Failure. Int. J. Linguist. Lit. Transl. 2024; 7 (9): 183–198. Publisher Full Text Aransa W: Statistical Machine Translation of the Arabic Dialect. Ph.D. thesis. University of Maine, doctoral school STIM, 2015. Arzu A, Issa T: An effect on cultural identity: Dialect. Procedia Soc. Behav. Sci. 2014; 143 : 555–562. Publisher Full Text Bamunusinghe K, Bamunusinghe S: The Importance of the Knowledge on Dialects for a Translator.2014. Reference Source Bassnett S: Translation studies. Routledge; 2013. Brown PF, Della Pietra SA, Della Pietra VJ, et al. : The mathematics of statistical machine translation: Parameter estimation. Comput. Linguist. 1993; 19 (2): 263–311. Cao Y, Kementchedjhieva Y, Cui R, et al. : Cultural Adaptation of Recipes. Trans. Assoc. Comput. Linguist. 2024; 12 : 80–99. Publisher Full Text Castellani B: Automatic generation of morpheme level reordering rules for Korean to English machine translation. MA thesis, Seoul National University; 2017. Çetin Ȍ, Duran A: A Comparative Analysis of the Performances of ChatGPT, DeepL, Google Translate and a Human Translator in Community Based Settings. Amasya Universitesi Sosyal Bilimler Dergisi. 2024; 9 (15): 120–173. Datta S: Investigating English-Language Dialect-Adjusted Models. Computer Science Senior Theses.2023; 11. Reference Source Daulay E: The social meaning of language and dialect. VISION. 2017; 12 (12). Deilen S, Garrido S, Lapshinova-Koltunski E, et al. : Using ChatGPT as a CAT Tool in Easy Language Translation. Proceedings of the Second Workshop on Text Simplification, Accessibility and Readability Associated with RANLP. 2023; pp. 1–10. Publisher Full Text Ding L, Wu D, Tao D: Improving neural machine translation by bidirectional training. arXiv preprint arXiv: 2109.07780. 2021. Durrani N, Al-Onaizan Y, Ittycheriah A: Improving Egyptian-to-English SMT by mapping Egyptian into MSA.International Conference on Intelligent Text Processing and Computational Linguistics.Berlin, Heidelberg: Springer Berlin Heidelberg; 2014; pp. 271–282. Publisher Full Text Elkaffash SM: Corpus-Based Quality Evaluation of Ar-En Neural Machine Translation: Google Translate as a Case Study. Master’s thesis, Hamad Bin Khalifa University (Qatar); 2020. Fairclough N: Analysing discourse. London: Routledge; 2003. Federici FM: Translating Dialects and Languages of Minorities: Challenges and Solutions. Die Deutsche Nationalbibliothek; 2011. Ferguson CA: Diglossia. Word. 1959; 15 : 325–340. Publisher Full Text Gill S, Kaur R: ChatGPT: Vision and Challenges. TCPS. Elsevier; 2023. Groves D, Dag S: Identification and analysis of post-editing patterns for MT. Proceedings of the Twelfth Machine Translation Summit, August 26–30, Ottawa. 2009; 429–436. Guo D, Zhu Q, Yang D, et al. : DeepSeek-Coder: When the Large Language Model Meets Programming- the Rise of Code Intelligence.2024. Reference Source Hofmann H, Sakti S, Isotani R, et al. : Sequence-based pronunciation modeling using a noisy-channel approach. Spoken Dialogue Systems for Ambient Environments: Second International Workshop on Spoken Dialogue Systems Technology, IWSDS 2010, Gotemba, Shizuoka, Japan, October 1-2, 2010. Proceedings. Berlin Heidelberg: Springer; 2010; pp. 156–162. Hutchins WJ: Machine translation: past, present, future. Chichester: Ellis Horwood; 1986. Jiang L, Jiang Y, Han L: The Potential of ChatGPT in Translation Evaluation: A Case Study of the Chinese-Portuguese Machine Translation. Casernos de Traduçao. 2024; 44 : 1–22. Publisher Full Text Jiao W, Wang W, Huang J, et al. : Is ChatGPT a good translator? Yes with GPT-4 as the engine.2023. Reference Source Joshi S: A Comprehensive Review of DeepSeek: Performance, Architecture and Capabilities.2025. Publisher Full Text Kadaoui K, Magdy SM, Waheed A, et al. : Tarjamat: Evaluation of bard and chatgpt on machine translation of ten arabic varieties. arXiv preprint arXiv:2308.03051. 2023. Koehn P: Statistical Machine Translation. Cambridge: Cambridge University Press; 2009. Koehn P, Och FJ, Marcu D: Statistical phrase-based translation. In Proceedings of the Joint Conference on Human Language Technologies and the Annual Meeting of the North Ameri can Chapter of the Association of Computational Linguistics (HLT-NAACL).2003. Reference Source Kong JW: Translating Dialects and Languages of Minorities: Challenges and Solutions. Review. Babel. 2013; 59 (1): 121–124. Federici, F. M. (ed.). (2011). Publisher Full Text Krings H: Repairing texts: empirical investigations of machine translation post-editing processes. Kent, OH: The Kent State University Press; 2001. Kumar Y, Gordon Z, Alabi O, et al. : ChatGPT Translation of Program Code for Image Sketch Abstraction. Appl. Sci. 2024; 14 (3): 992. Publisher Full Text Lee TK: Artificial intelligence and Posthumanist translation: ChatGPT vs the translator. Appl. Linguist. Rev. 2023; 15 : 2351–2372. Publisher Full Text Lilli S: ChatGPT-4 and Italian dialects: assessing linguistic competence. Umanistica Digit. 2023; 16 : 235–263. Macken L: Machine Translation Meets Large Language Models: Evaluating ChatGPT’s Ability to Aautomatically Post-Edit Literary Texts. Proceedings of the 1st Workshop on Creative-Text Translation and Technology. 2024; pp. 65–81. Neubig G, Akita Y, Mori S, et al. : Improved statistical models for SMT-based speaking style transformation. 2010 IEEE International Conference on Acoustics, Speech and Signal Processing. IEEE; 2010, March; pp. 5206–5209. Newmark P: A textbook of translation. New York: Prentice Hall; 1988. Peng K, Ding L, Zhong Q, et al. : Towards making the most of chatgpt for machine translation. arXiv preprint arXiv: 2303.13780. 2023. Peng Y, Chen Q, Shih G: DeepSeek is Open-Access and the Next AI Disrupter for Radiology. Radiol. Adv. 2025; 2 : 1. Publisher Full Text Postigo M: ChatGPT and MT-Systems: Advantages and Limitations when Translating English to Spanish and Portuguese. Vol. 28 . . Lengua Y Habla; 2024. Puppel M, Borg C: Evaluating ChatGPT’s Performance in Creative Text Translation for Communication: A Case Study from English into German. Media and Intercultural Communication: A Multidisciplinary Journal. 2024; 3 (1): 1–27. Publisher Full Text Saito D, Watanabe S, Nakamura A, et al. : Statistical voice conversion based on noisy channel model. IEEE Trans. Audio Speech Lang. Process. 2012; 20 (6): 1784–1794. Publisher Full Text Salloum W, Habash N: Elissa: A dialectal to standard Arabic machine translation system. Proceedings of COLING 2012: Demonstration papers. 2012; pp. 385–392. Sawaf H: Arabic dialect handling in hybrid machine translation. Pro-ceedings of the Conference of the Association for Machine Translation in the Americas (AMTA). Denver, Colorado; 2010. Shormani MQ: An introduction to English syntax. A generative approach. LAP Lambert Academic Publishing; 2013. Shormani MQ: Vocatives in Yemeni (ibbi) Arabic: Functions, types and approach. J. Semit. Stud. 2019; 64 (1): 221–250. Publisher Full Text Shormani MQ: Does culture translate? Evidence from translating proverbs. Babel, John Benjamins. 2020; 66 (6): 902–927. Publisher Full Text Shormani MQ: Can ChatGPT capture swearing nuances? Evidence from translating Arabic oaths.2024a. Reference Source Shormani MQ: Linguistics contribution to artificial intelligence Where this contribution lies.2024b. Publisher Full Text Shormani MQ: Introducing minimalism: A parametric variation. Lincom Europa Press; 2024c. Shormani MQ: Non-native speakers of English or ChatGPT: Who thinks better? F1000Res. 2025a; 14 . Publisher Full Text Shormani MQ: AI translation and culture-based expressions. A lecture given at AlQalam University, held on 16/01/2025.2025b. Shormani MQ, Al-samki AA: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point.2025. Shormani MQ, Alfahd A: Artificial Intelligence or Human: The use of ChatGPT in the academic translation for religious texts (To appear in Sage Open).2025. Shormani MQ, AlSohbani YA: Artificial intelligence contribution to translation industry: looking back and forward. Discov. Artif. Intell. 2025; 5 : 389. Publisher Full Text Shormani MQ, Watson JC, Dickins J: Poems from Ibb and Hadramawt. In Morano R, Watson J, Dickins, editors. Yemeni Poetry on the Frontline: love and conflict. Routledge; 2025; pp. 14–31. Sindhuja R: Translation Theory and Practice.2021. Reference Source Siu SC: ChatGPT and GPT-4 for professional translators: exploring the potential of large language models in translation. Preprint. 2023; 1–36. Reference Source Vaswani A, Shazeer N, Parmar N, et al. : Attention is all you need. Adv. Neural Inf. Proces. Syst. 2017; 30 . Wang C, Kantarcioglu M: A Review of DeepSeek Models’ Key Innovative Techniques. arXiv:2503.11486v1. 2025. Wu Y, Schuster M, Chen Z, et al. : Google’s neural machine translation system: Bridging the gap between human and machine translation. arXiv preprint arXiv:1609.08144. 2016. Zan C, Peng K, Ding L, et al. : Vega-mt: The jd explore academy translation system for wmt22. arXiv preprint arXiv: 2209.09444. 2022 Zbib R, Malchiodi E, Devlin J, et al. : Machine translation of Arabic dialects. Proceedings of the 2012 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 2012: 49–59. Comments on this article Comments (0) Version 2 VERSION 2 PUBLISHED 16 Jul 2025 ADD YOUR COMMENT Comment Author details Author details 1 Department of English Studies, Ibb University, Ibb, Ibb Governorate, Yemen Mohammed Q. Shormani Roles: Conceptualization, Formal Analysis, Methodology, Software, Writing – Original Draft Preparation, Writing – Review & Editing Alia. Ali Al-Samki Roles: Data Curation, Resources, Validation, Writing – Original Draft Preparation, Writing – Review & Editing Competing interests No competing interests were disclosed. Grant information The author(s) declared that no grants were involved in supporting this work. Article Versions (2) version 2 Revised Published: 16 Jan 2026, 14:694 https://doi.org/10.12688/f1000research.165879.2 version 1 Published: 16 Jul 2025, 14:694 https://doi.org/10.12688/f1000research.165879.1 Copyright © 2026 Shormani MQ and Al-Samki AA. This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. Download Export To Sciwheel Bibtex EndNote ProCite Ref. Manager (RIS) Sente metrics Views Downloads F1000Research - - PubMed Central info_outline Data from PMC are received and updated monthly. - - Citations open_in_new 0 open_in_new 0 open_in_new SEE MORE DETAILS CITE how to cite this article Shormani MQ and Al-Samki AA. Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.12688/f1000research.165879.2 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS track receive updates on this article Track an article to receive email alerts on any updates to this article. TRACK THIS ARTICLE Share Open Peer Review Current Reviewer Status: ? Key to Reviewer Statuses VIEW HIDE Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Version 2 VERSION 2 PUBLISHED 16 Jan 2026 Revised Views 0 Cite How to cite this report: Al-Khalifa HS. Reviewer Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.194488.r460322 ) The direct URL for this report is: https://f1000research.com/articles/14-694/v2#referee-response-460322 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 06 Mar 2026 Hend S Al-Khalifa , King Saud University, Riyadh, Saudi Arabia Not Approved VIEWS 0 https://doi.org/10.5256/f1000research.194488.r460322 Here is the refined review rewritten in coherent paragraph form: The introduction should clearly and explicitly state the research objectives and research questions at the beginning of the manuscript, rather than introducing them later in the paper. Presenting these ... Continue reading READ ALL Here is the refined review rewritten in coherent paragraph form: The introduction should clearly and explicitly state the research objectives and research questions at the beginning of the manuscript, rather than introducing them later in the paper. Presenting these elements upfront would help readers better understand the scope and motivation of the study from the outset. Section 2 contains only one subsection (2.1), making this subdivision unnecessary in the absence of additional subsections. Furthermore, Section 2.1 is overly verbose and includes repeated information. Much of this content could be streamlined and more effectively presented using figures and tables. Similarly, Section 3 also contains only one subsection (3.1), which is unnecessary given the lack of further subdivisions. In addition, although this section is titled “Translating Dialects,” its content primarily focuses on translation between languages rather than dialectal translation. This mismatch reduces the clarity and coherence of the section. It would be more appropriate to include a dedicated section that specifically addresses interlanguage translation. The exclusive use of ChatGPT and DeepSeek in the study is not sufficiently justified. Evaluating a broader range of large language models would strengthen the empirical foundation of the work and enhance its contribution. Moreover, the dataset used in the study is very limited in size and does not appear to support a substantial scientific contribution. The criteria for selecting the categories and lexical items are also not clearly explained, which further weakens the methodological rigor. The methodological procedure lacks sufficient detail and transparency. In particular, the prompts used for generating translations are not described. It remains unclear whether zero-shot or few-shot prompting strategies were employed and whether the prompts were formulated in English or Arabic. In addition, the evaluation process is insufficiently documented. The number of human evaluators is not reported, and no information is provided regarding inter-annotator agreement, making it difficult to assess the reliability of the results. The discussion section largely reiterates the findings already presented in the Results section, rather than providing deeper interpretation, critical analysis, or theoretical engagement. Similarly, the section on “Linguistic Subtleties” is superficial and would benefit from a more rigorous engagement with established linguistic theories and relevant scholarly literature. The conclusion of the paper is largely generic and reflects observations that could apply to most studies on low-resource or dialectal Arabic. As a result, it does not sufficiently highlight the specific contributions or implications of the present study. Overall, the manuscript is excessively verbose, with repeated details appearing across multiple sections and subsections that often stand alone. This structural imbalance negatively affects the coherence, readability, and flow of the paper. Greater concision and more careful organization would substantially improve the quality and impact of the manuscript. Is the work clearly and accurately presented and does it cite the current literature? Partly Is the study design appropriate and is the work technically sound? No Are sufficient details of methods and analysis provided to allow replication by others? No If applicable, is the statistical analysis and its interpretation appropriate? Not applicable Are all the source data underlying the results available to ensure full reproducibility? Partly Are the conclusions drawn adequately supported by the results? Partly Competing Interests: No competing interests were disclosed. Reviewer Expertise: Arabic NLP I confirm that I have read this submission and believe that I have an appropriate level of expertise to state that I do not consider it to be of an acceptable scientific standard, for reasons outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Al-Khalifa HS. Reviewer Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.194488.r460322 ) The direct URL for this report is: https://f1000research.com/articles/14-694/v2#referee-response-460322 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Ding L. Reviewer Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.194488.r450599 ) The direct URL for this report is: https://f1000research.com/articles/14-694/v2#referee-response-450599 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 20 Jan 2026 Liang Ding , The University of Sydney, Sydney, Australia Approved VIEWS 0 https://doi.org/10.5256/f1000research.194488.r450599 Thank the authors for ... Continue reading READ ALL Thank the authors for addressing most of my concerns. Competing Interests: No competing interests were disclosed. Reviewer Expertise: natural language processing, machine learning, large language models, machine translation I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Ding L. Reviewer Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.194488.r450599 ) The direct URL for this report is: https://f1000research.com/articles/14-694/v2#referee-response-450599 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Version 1 VERSION 1 PUBLISHED 16 Jul 2025 Views 0 Cite How to cite this report: Ding L. Reviewer Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.182657.r403515 ) The direct URL for this report is: https://f1000research.com/articles/14-694/v1#referee-response-403515 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 25 Sep 2025 Liang Ding , The University of Sydney, Sydney, Australia Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.182657.r403515 This study evaluates the performance of two large language models, ChatGPT-4o and DeepSeek v3, on the task of translating 50 culturally specific dialectical terms from Yemeni Sana'ani Arabic (YSA) into English. The authors perform a qualitative and quantitative analysis, ... Continue reading READ ALL This study evaluates the performance of two large language models, ChatGPT-4o and DeepSeek v3, on the task of translating 50 culturally specific dialectical terms from Yemeni Sana'ani Arabic (YSA) into English. The authors perform a qualitative and quantitative analysis, comparing the models' outputs against a human-provided translation. The core findings indicate that both models struggle significantly with the nuances of the YSA dialect, frequently defaulting to Standard Arabic (SA) interpretations or producing literal, contextually incorrect translations. The study concludes that ChatGPT performs marginally better than DeepSeek, but both are inadequate for reliable dialect translation, highlighting a critical need for dialect-aware data and model development. Pros: 1) The translation of low-resource and non-standard dialects is a critical frontier for machine translation. This work addresses a significant gap by focusing on YSA, an understudied variety, providing valuable initial insights into the limitations of current state-of-the-art LLMs. 2) To my knowledge, this is one of the first academic papers to benchmark the DeepSeek model against ChatGPT on a dialectal translation task, making a timely contribution to the comparative analysis of emerging LLMs. 3) The paper excels in its qualitative discussion. The term-by-term breakdown in Table 2 and the subsequent discussion of cultural, linguistic, and contextual subtleties (Sections 5.2 and 6) provide clear, compelling evidence of the models' failures and the reasons behind them (e.g., mistaking شركة for 'company' instead of 'meat'). Cons: The paper, while valuable for its topic, is constrained by significant experimental and contextual limitations that temper its conclusions. 1) The analysis relies solely on the authors' classification of translations as "correct," "incorrect," or "appropriate". This is highly subjective and lacks the rigor of standard MT evaluation. State-of-the-art research has moved beyond lexical-overlap metrics like BLEU to model-based metrics such as COMET, which better capture semantic fidelity. The absence of such metrics makes the quantitative claims (e.g., 34% correct for ChatGPT) difficult to verify and compare against other work. 2) The study's methodology does not reflect the current best practices for evaluating LLMs in translation tasks. No Prompt Engineering: The paper fails to specify the prompts used, but the results suggest a simple, zero-shot directive (e.g., "Translate X"). This is a critical flaw. It is well-documented that LLM performance is highly sensitive to prompting strategies. The models were not given a fair chance to perform well. Term-Level Translation: The evaluation is conducted on isolated terms. This is not representative of real-world usage and ignores the crucial role of context in disambiguation. Translating the terms within full sentences would have provided a more realistic and challenging test. 3) The recommendations for AI developers are generic (e.g., "expand the dialectal data coverage"). The paper would be substantially strengthened by engaging with more specific, high-impact research in machine translation to offer more sophisticated solutions. To address these shortcomings, the authors should incorporate and cite the following highly relevant papers : Peng, K. et al. (2023) [1]. Towards Making the Most of ChatGPT for Machine Translation : This paper is essential reading. It explicitly demonstrates that simple prompting limits ChatGPT's translation ability. The authors should have experimented with the Task-Specific Prompts (TSP) and Domain-Specific Prompts (DSP) are proposed in this work to see if defining the task more clearly ("You are a machine translation system translating the Yemeni Sana'ani dialect") improves performance. Furthermore, this paper advocates for using COMET as a primary metric, a practice this study should adopt. Zan, C. et al. (2022) [2]. Vega-MT: The JD Explore Academy Translation System for WMT22 : While focused on a traditional NMT system, this paper offers a blueprint for building high-quality multilingual models. Its concept of "multi-directional pretraining" —using data from all language pairs to exploit common knowledge—is a concrete technical strategy that goes beyond the paper's generic recommendation to simply "add more data". The authors could propose a "multi-dialectal" pre-training approach inspired by this work as a specific path forward. Ding, L. et al. (2021) [3]. Improving Neural Machine Translation by Bidirectional Training: This paper introduces Bidirectional Training (BiT) , a simple yet effective strategy of pre-training a model on both src -> tgt and tgt -> src data simultaneously. This method was shown to improve performance, especially in low-resource settings. Suggesting a BiT-based fine-tuning approach (using both YSA-to-English and English-to-YSA data) would be a specific, testable hypothesis for improving the models' grasp of YSA's linguistic structure and improving bilingual alignment. Ref: [1] Peng et al. Towards Making the Most of ChatGPT for Machine Translation. In Findings of EMNLP 2023. [2] Zan et al., Vega-MT: The JD Explore Academy Translation System for WMT22. In WMT 2022. [3] Ding et al., Improving Neural Machine Translation by Bidirectional Training. In EMNLP 2021. Is the work clearly and accurately presented and does it cite the current literature? Yes Is the study design appropriate and is the work technically sound? Yes Are sufficient details of methods and analysis provided to allow replication by others? Yes If applicable, is the statistical analysis and its interpretation appropriate? Yes Are all the source data underlying the results available to ensure full reproducibility? No source data required Are the conclusions drawn adequately supported by the results? Yes Competing Interests: No competing interests were disclosed. Reviewer Expertise: natural language processing, machine learning, large language models, machine translation I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Ding L. Reviewer Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.182657.r403515 ) The direct URL for this report is: https://f1000research.com/articles/14-694/v1#referee-response-403515 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Author Response 02 Jan 2026 Mohammed Q. Shormani , English Studies, Ibb University, Ibb, Yemen 02 Jan 2026 Author Response Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find ... Continue reading Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point as follows. This study evaluates the performance of two large language models, ChatGPT-4o and DeepSeek v3, on the task of translating 50 culturally specific dialectical terms from Yemeni Sana'ani Arabic (YSA) into English. The authors perform a qualitative and quantitative analysis, comparing the models' outputs against a human-provided translation. The core findings indicate that both models struggle significantly with the nuances of the YSA dialect, frequently defaulting to Standard Arabic (SA) interpretations or producing literal, contextually incorrect translations. The study concludes that ChatGPT performs marginally better than DeepSeek, but both are inadequate for reliable dialect translation, highlighting a critical need for dialect-aware data and model development. Response Thank you very much for your valuable remark. 1)The translation of low-resource and non-standard dialects is a critical frontier for machine translation. This work addresses a significant gap by focusing on YSA, an understudied variety, providing valuable initial insights into the limitations of current state-of-the-art LLMs. Response Thank you very much for your valuable remark. 2) To my knowledge, this is one of the first academic papers to benchmark the DeepSeek model against ChatGPT on a dialectal translation task, making a timely contribution to the comparative analysis of emerging LLMs. Response Thank you very much for your valuable remark. 3) The paper excels in its qualitative discussion. The term-by-term breakdown in Table 2 and the subsequent discussion of cultural, linguistic, and contextual subtleties (Sections 5.2 and 6) provide clear, compelling evidence of the models' failures and the reasons behind them (e.g., mistaking شركة for 'company' instead of 'meat'). Response Thank you very much for your valuable remark. The analysis relies solely on the authors' classification of translations as "correct," "incorrect," or "appropriate". This is highly subjective and lacks the rigor of standard MT evaluation. State-of-the-art research has moved beyond lexical-overlap metrics like BLEU to model-based metrics such as COMET, which better capture semantic fidelity. The absence of such metrics makes the quantitative claims (e.g., 34% correct for ChatGPT) difficult to verify and compare against other work. Response Thank you very much for your valuable comment. Please note that our use of the categories correct, incorrect, and appropriate was intentionally motivated by the nature of the task: translating Sana’ni Arabic, a dialect for which no standardized reference corpora or validated automatic evaluation benchmarks currently exist. As a result, commonly used model-based metrics such as COMET—which require high-quality reference translations and have been trained predominantly on Standard Arabic and high-resource language pairs—are not yet reliable indicators of translation quality for this dialect. I would like also to clarify that our quantitative results as indicative rather than absolute. I have also developed this part by incoporating the following text " We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses." 2) The study's methodology does not reflect the current best practices for evaluating LLMs in translation tasks. No Prompt Engineering: The paper fails to specify the prompts used, but the results suggest a simple, zero-shot directive (e.g., "Translate X"). This is a critical flaw. It is well-documented that LLM performance is highly sensitive to prompting strategies. The models were not given a fair chance to perform well. Term-Level Translation: The evaluation is conducted on isolated terms. This is not representative of real-world usage and ignores the crucial role of context in disambiguation. Translating the terms within full sentences would have provided a more realistic and challenging test. Response Thank you very much for your valuable comment. I have addressed this aspect, and acknowledged it as one limitation of the study. Please section 7. The recommendations for AI developers are generic (e.g., "expand the dialectal data coverage"). The paper would be substantially strengthened by engaging with more specific, high-impact research in machine translation to offer more sophisticated solutions....... Response Thank you very much for your valuable comment. I have addressed these outstanding points and referred to these references. Please see section 7 [1] Peng et al. Towards Making the Most of ChatGPT for Machine Translation. In Findings of EMNLP 2023. [2] Zan et al., Vega-MT: The JD Explore Academy Translation System for WMT22. In WMT 2022. [3] Ding et al., Improving Neural Machine Translation by Bidirectional Training. In EMNLP 2021. Finally, thank you very much for your valuable comments and for the efforts you exerted to improving the article/ Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point as follows. This study evaluates the performance of two large language models, ChatGPT-4o and DeepSeek v3, on the task of translating 50 culturally specific dialectical terms from Yemeni Sana'ani Arabic (YSA) into English. The authors perform a qualitative and quantitative analysis, comparing the models' outputs against a human-provided translation. The core findings indicate that both models struggle significantly with the nuances of the YSA dialect, frequently defaulting to Standard Arabic (SA) interpretations or producing literal, contextually incorrect translations. The study concludes that ChatGPT performs marginally better than DeepSeek, but both are inadequate for reliable dialect translation, highlighting a critical need for dialect-aware data and model development. Response Thank you very much for your valuable remark. 1)The translation of low-resource and non-standard dialects is a critical frontier for machine translation. This work addresses a significant gap by focusing on YSA, an understudied variety, providing valuable initial insights into the limitations of current state-of-the-art LLMs. Response Thank you very much for your valuable remark. 2) To my knowledge, this is one of the first academic papers to benchmark the DeepSeek model against ChatGPT on a dialectal translation task, making a timely contribution to the comparative analysis of emerging LLMs. Response Thank you very much for your valuable remark. 3) The paper excels in its qualitative discussion. The term-by-term breakdown in Table 2 and the subsequent discussion of cultural, linguistic, and contextual subtleties (Sections 5.2 and 6) provide clear, compelling evidence of the models' failures and the reasons behind them (e.g., mistaking شركة for 'company' instead of 'meat'). Response Thank you very much for your valuable remark. The analysis relies solely on the authors' classification of translations as "correct," "incorrect," or "appropriate". This is highly subjective and lacks the rigor of standard MT evaluation. State-of-the-art research has moved beyond lexical-overlap metrics like BLEU to model-based metrics such as COMET, which better capture semantic fidelity. The absence of such metrics makes the quantitative claims (e.g., 34% correct for ChatGPT) difficult to verify and compare against other work. Response Thank you very much for your valuable comment. Please note that our use of the categories correct, incorrect, and appropriate was intentionally motivated by the nature of the task: translating Sana’ni Arabic, a dialect for which no standardized reference corpora or validated automatic evaluation benchmarks currently exist. As a result, commonly used model-based metrics such as COMET—which require high-quality reference translations and have been trained predominantly on Standard Arabic and high-resource language pairs—are not yet reliable indicators of translation quality for this dialect. I would like also to clarify that our quantitative results as indicative rather than absolute. I have also developed this part by incoporating the following text " We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses." 2) The study's methodology does not reflect the current best practices for evaluating LLMs in translation tasks. No Prompt Engineering: The paper fails to specify the prompts used, but the results suggest a simple, zero-shot directive (e.g., "Translate X"). This is a critical flaw. It is well-documented that LLM performance is highly sensitive to prompting strategies. The models were not given a fair chance to perform well. Term-Level Translation: The evaluation is conducted on isolated terms. This is not representative of real-world usage and ignores the crucial role of context in disambiguation. Translating the terms within full sentences would have provided a more realistic and challenging test. Response Thank you very much for your valuable comment. I have addressed this aspect, and acknowledged it as one limitation of the study. Please section 7. The recommendations for AI developers are generic (e.g., "expand the dialectal data coverage"). The paper would be substantially strengthened by engaging with more specific, high-impact research in machine translation to offer more sophisticated solutions....... Response Thank you very much for your valuable comment. I have addressed these outstanding points and referred to these references. Please see section 7 [1] Peng et al. Towards Making the Most of ChatGPT for Machine Translation. In Findings of EMNLP 2023. [2] Zan et al., Vega-MT: The JD Explore Academy Translation System for WMT22. In WMT 2022. [3] Ding et al., Improving Neural Machine Translation by Bidirectional Training. In EMNLP 2021. Finally, thank you very much for your valuable comments and for the efforts you exerted to improving the article/ Competing Interests: No competing interest to disclose. Close Report a concern Respond or Comment COMMENTS ON THIS REPORT Author Response 02 Jan 2026 Mohammed Q. Shormani , English Studies, Ibb University, Ibb, Yemen 02 Jan 2026 Author Response Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find ... Continue reading Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point as follows. This study evaluates the performance of two large language models, ChatGPT-4o and DeepSeek v3, on the task of translating 50 culturally specific dialectical terms from Yemeni Sana'ani Arabic (YSA) into English. The authors perform a qualitative and quantitative analysis, comparing the models' outputs against a human-provided translation. The core findings indicate that both models struggle significantly with the nuances of the YSA dialect, frequently defaulting to Standard Arabic (SA) interpretations or producing literal, contextually incorrect translations. The study concludes that ChatGPT performs marginally better than DeepSeek, but both are inadequate for reliable dialect translation, highlighting a critical need for dialect-aware data and model development. Response Thank you very much for your valuable remark. 1)The translation of low-resource and non-standard dialects is a critical frontier for machine translation. This work addresses a significant gap by focusing on YSA, an understudied variety, providing valuable initial insights into the limitations of current state-of-the-art LLMs. Response Thank you very much for your valuable remark. 2) To my knowledge, this is one of the first academic papers to benchmark the DeepSeek model against ChatGPT on a dialectal translation task, making a timely contribution to the comparative analysis of emerging LLMs. Response Thank you very much for your valuable remark. 3) The paper excels in its qualitative discussion. The term-by-term breakdown in Table 2 and the subsequent discussion of cultural, linguistic, and contextual subtleties (Sections 5.2 and 6) provide clear, compelling evidence of the models' failures and the reasons behind them (e.g., mistaking شركة for 'company' instead of 'meat'). Response Thank you very much for your valuable remark. The analysis relies solely on the authors' classification of translations as "correct," "incorrect," or "appropriate". This is highly subjective and lacks the rigor of standard MT evaluation. State-of-the-art research has moved beyond lexical-overlap metrics like BLEU to model-based metrics such as COMET, which better capture semantic fidelity. The absence of such metrics makes the quantitative claims (e.g., 34% correct for ChatGPT) difficult to verify and compare against other work. Response Thank you very much for your valuable comment. Please note that our use of the categories correct, incorrect, and appropriate was intentionally motivated by the nature of the task: translating Sana’ni Arabic, a dialect for which no standardized reference corpora or validated automatic evaluation benchmarks currently exist. As a result, commonly used model-based metrics such as COMET—which require high-quality reference translations and have been trained predominantly on Standard Arabic and high-resource language pairs—are not yet reliable indicators of translation quality for this dialect. I would like also to clarify that our quantitative results as indicative rather than absolute. I have also developed this part by incoporating the following text " We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses." 2) The study's methodology does not reflect the current best practices for evaluating LLMs in translation tasks. No Prompt Engineering: The paper fails to specify the prompts used, but the results suggest a simple, zero-shot directive (e.g., "Translate X"). This is a critical flaw. It is well-documented that LLM performance is highly sensitive to prompting strategies. The models were not given a fair chance to perform well. Term-Level Translation: The evaluation is conducted on isolated terms. This is not representative of real-world usage and ignores the crucial role of context in disambiguation. Translating the terms within full sentences would have provided a more realistic and challenging test. Response Thank you very much for your valuable comment. I have addressed this aspect, and acknowledged it as one limitation of the study. Please section 7. The recommendations for AI developers are generic (e.g., "expand the dialectal data coverage"). The paper would be substantially strengthened by engaging with more specific, high-impact research in machine translation to offer more sophisticated solutions....... Response Thank you very much for your valuable comment. I have addressed these outstanding points and referred to these references. Please see section 7 [1] Peng et al. Towards Making the Most of ChatGPT for Machine Translation. In Findings of EMNLP 2023. [2] Zan et al., Vega-MT: The JD Explore Academy Translation System for WMT22. In WMT 2022. [3] Ding et al., Improving Neural Machine Translation by Bidirectional Training. In EMNLP 2021. Finally, thank you very much for your valuable comments and for the efforts you exerted to improving the article/ Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point as follows. This study evaluates the performance of two large language models, ChatGPT-4o and DeepSeek v3, on the task of translating 50 culturally specific dialectical terms from Yemeni Sana'ani Arabic (YSA) into English. The authors perform a qualitative and quantitative analysis, comparing the models' outputs against a human-provided translation. The core findings indicate that both models struggle significantly with the nuances of the YSA dialect, frequently defaulting to Standard Arabic (SA) interpretations or producing literal, contextually incorrect translations. The study concludes that ChatGPT performs marginally better than DeepSeek, but both are inadequate for reliable dialect translation, highlighting a critical need for dialect-aware data and model development. Response Thank you very much for your valuable remark. 1)The translation of low-resource and non-standard dialects is a critical frontier for machine translation. This work addresses a significant gap by focusing on YSA, an understudied variety, providing valuable initial insights into the limitations of current state-of-the-art LLMs. Response Thank you very much for your valuable remark. 2) To my knowledge, this is one of the first academic papers to benchmark the DeepSeek model against ChatGPT on a dialectal translation task, making a timely contribution to the comparative analysis of emerging LLMs. Response Thank you very much for your valuable remark. 3) The paper excels in its qualitative discussion. The term-by-term breakdown in Table 2 and the subsequent discussion of cultural, linguistic, and contextual subtleties (Sections 5.2 and 6) provide clear, compelling evidence of the models' failures and the reasons behind them (e.g., mistaking شركة for 'company' instead of 'meat'). Response Thank you very much for your valuable remark. The analysis relies solely on the authors' classification of translations as "correct," "incorrect," or "appropriate". This is highly subjective and lacks the rigor of standard MT evaluation. State-of-the-art research has moved beyond lexical-overlap metrics like BLEU to model-based metrics such as COMET, which better capture semantic fidelity. The absence of such metrics makes the quantitative claims (e.g., 34% correct for ChatGPT) difficult to verify and compare against other work. Response Thank you very much for your valuable comment. Please note that our use of the categories correct, incorrect, and appropriate was intentionally motivated by the nature of the task: translating Sana’ni Arabic, a dialect for which no standardized reference corpora or validated automatic evaluation benchmarks currently exist. As a result, commonly used model-based metrics such as COMET—which require high-quality reference translations and have been trained predominantly on Standard Arabic and high-resource language pairs—are not yet reliable indicators of translation quality for this dialect. I would like also to clarify that our quantitative results as indicative rather than absolute. I have also developed this part by incoporating the following text " We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses." 2) The study's methodology does not reflect the current best practices for evaluating LLMs in translation tasks. No Prompt Engineering: The paper fails to specify the prompts used, but the results suggest a simple, zero-shot directive (e.g., "Translate X"). This is a critical flaw. It is well-documented that LLM performance is highly sensitive to prompting strategies. The models were not given a fair chance to perform well. Term-Level Translation: The evaluation is conducted on isolated terms. This is not representative of real-world usage and ignores the crucial role of context in disambiguation. Translating the terms within full sentences would have provided a more realistic and challenging test. Response Thank you very much for your valuable comment. I have addressed this aspect, and acknowledged it as one limitation of the study. Please section 7. The recommendations for AI developers are generic (e.g., "expand the dialectal data coverage"). The paper would be substantially strengthened by engaging with more specific, high-impact research in machine translation to offer more sophisticated solutions....... Response Thank you very much for your valuable comment. I have addressed these outstanding points and referred to these references. Please see section 7 [1] Peng et al. Towards Making the Most of ChatGPT for Machine Translation. In Findings of EMNLP 2023. [2] Zan et al., Vega-MT: The JD Explore Academy Translation System for WMT22. In WMT 2022. [3] Ding et al., Improving Neural Machine Translation by Bidirectional Training. In EMNLP 2021. Finally, thank you very much for your valuable comments and for the efforts you exerted to improving the article/ Competing Interests: No competing interest to disclose. Close Report a concern COMMENT ON THIS REPORT Views 0 Cite How to cite this report: DINH CT. Reviewer Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.182657.r409822 ) The direct URL for this report is: https://f1000research.com/articles/14-694/v1#referee-response-409822 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 24 Sep 2025 Cao-Tuong DINH , FPT University, Can Tho city, Vietnam Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.182657.r409822 Recommendation: Major revisions Required For author and editor Thank you for the opportunity to review this paper. I appreciate the effort that went into this research and the attempt to apply computational methods to analyze diplomatic ... Continue reading READ ALL Recommendation: Major revisions Required For author and editor Thank you for the opportunity to review this paper. I appreciate the effort that went into this research and the attempt to apply computational methods to analyze diplomatic discourse. The topic is timely and valuable for MT/LLM evaluation beyond standard Arabic. The manuscript is generally clear, and the dataset is shared. However, methodological transparency and rigor need strengthening (sampling, prompting, validation, and statistics). A few internal inconsistencies and copy-editing issues also require attention. With moderate revisions, the paper will provide a useful exploratory baseline, hence Approved with major revision . However, as a researcher, I have some concerns as below: 1) Clarity & literature context Mostly clear, with a good high-level MT background. Literature is broadly cited, but several claims (e.g., model capabilities, architecture details, and comparative statements) would benefit from primary/technical citations and tighter focus on dialect MT work. In particular, these points need improvement: Tighten the background to emphasize prior work on Arabic dialect MT and dialectal evaluation, and separate general MT history from directly relevant work. Add citations for specific DeepSeek details you reference (Mixture-of-Experts, training corpora) and for Arabic dialect evaluation benchmarks/tools where applicable. Copy-edit for typos/wording (examples in Minor comments). 2) Study design & academic merit: The exploratory design (50 terms) is a reasonable pilot, but sampling and gold-standard construction need more rigor to support conclusions. However, these points need modification: Sampling: Clarify how the 50 terms were selected (criteria, sources, representativeness across categories; frequency in real usage). Consider expanding to include contextualized uses (sentences) alongside isolated terms to reduce ambiguity. Gold standard: Specify the annotation protocol (number of native speakers, expertise, independence, adjudication process). Report inter-annotator agreement (e.g., Cohen’s κ) if multiple raters were used or describe how disagreements were resolved. Task framing: Define what counts as “correct,” “appropriate,” and “incorrect” a priori , with examples. Consider an error taxonomy (literal, SA-bias, cultural connotation loss, POS confusion, etc.). 3) Methods detail & reproducibility: Important details are currently missing for full replication. In particular, these aspects need to be clarified: Model versions & dates: Report exact model versions/variants, query timestamps (LLMs change over time), and any API/app settings. Prompts: Publish the exact prompts, instructions (e.g., “translate to English; provide multiple senses?”), temperature/decoding parameters, number of attempts/retries, and whether any post-processing was applied. Text normalization: Specify Unicode normalization, diacritic handling, tokenization, and whether Arabic script was normalized before translation. 4) Statistics & interpretation: Counts and percentages are reported, but statistical comparison is minimal. Below are suggestions for improvement: Add 95% CIs for accuracy/“correct” rates by category and overall. For paired categorical outcomes (ChatGPT vs DeepSeek on the same items), apply a McNemar test (or exact variant) to test whether differences are statistically significant. Report per-category performance with CIs and consider a simple mixed-effects model (random intercept for term) to account for item variability. If you keep “appropriate” as a middle category, consider ordinal models or report separate binary analyses (correct vs not; correct+appropriate vs incorrect). 5) Conclusions vs results: Broadly aligned, but some statements verge on over-generalization given sample size and item selection. Below are suggestions for improvement: Temper general claims about “SA bias” and cross-dialect performance; frame as evidence from this 50-term sample and invite replication on larger, balanced sets. Highlight that performance may change with contextual sentences and few-shot prompting; consider adding a small follow-up experiment (even in supplementary) to show sensitivity to prompt design. 6) Minor comments (clarity, style, presentation) Terminology consistency: Use “YSA” consistently (a few instances seem to vary). Typos & phrasing (examples): “concisderable” → considerable; “corp” → crops (re: silo); “wording”/“orthograpgical” → orthographical; ensure consistent capitalization of model names and sections. Is the work clearly and accurately presented and does it cite the current literature? Partly Is the study design appropriate and is the work technically sound? Partly Are sufficient details of methods and analysis provided to allow replication by others? Partly If applicable, is the statistical analysis and its interpretation appropriate? Partly Are all the source data underlying the results available to ensure full reproducibility? Partly Are the conclusions drawn adequately supported by the results? Partly Competing Interests: No competing interests were disclosed. Reviewer Expertise: self-regulated learning, EMI, technology-based teaching & learning in higher education I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT DINH CT. Reviewer Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.182657.r409822 ) The direct URL for this report is: https://f1000research.com/articles/14-694/v1#referee-response-409822 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Author Response 02 Jan 2026 Mohammed Q. Shormani , English Studies, Ibb University, Ibb, Yemen 02 Jan 2026 Author Response Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find ... Continue reading Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point below. Thank you for the opportunity to review this paper. I appreciate the effort that went into this research and the attempt to apply computational methods to analyze diplomatic discourse. The topic is timely and valuable for MT/LLM evaluation beyond standard Arabic. The manuscript is generally clear, and the dataset is shared. Response Thank you very much for your valuable remark. Mostly clear, with a good high-level MT background. Literature is broadly cited, but several claims (e.g., model capabilities, architecture details, and comparative statements) would benefit from primary/technical citations and tighter focus on dialect MT work. In particular, these points need improvement: Tighten the background to emphasize prior work on Arabic dialect MT and dialectal evaluation, and separate general MT history from directly relevant work. Add citations for specific DeepSeek details you reference (Mixture-of-Experts, training corpora) and for Arabic dialect evaluation benchmarks/tools where applicable. Copy-edit for typos/wording (examples in Minor comments) Response Thank you very much for your valuable comment. I have addressed these aspects, adding a subsection dubbed as "3.1. Translating Arabic dialects", and added several references as recommended by you and the other reviewer. The exploratory design (50 terms) is a reasonable pilot, but sampling and gold-standard construction need more rigor to support conclusions. However, these points need modification: Sampling: Clarify how the 50 terms were selected (criteria, sources, representativeness across categories; frequency in real usage). Consider expanding to include contextualized uses (sentences) alongside isolated terms to reduce ambiguity. Gold standard: Specify the annotation protocol (number of native speakers, expertise, independence, adjudication process). Report inter-annotator agreement (e.g., Cohen’s κ) if multiple raters were used or describe how disagreements were resolved. Task framing: Define what counts as “correct,” “appropriate,” and “incorrect” a priori , with examples. Consider an error taxonomy (literal, SA-bias, cultural connotation loss, POS confusion, etc.). Response Thank you very much for your valuable comment. I have addressed these aspects, detailing the criteria adopted, defining "what counts as “correct,” “appropriate,” and “incorrect”, and other related aspects. For example, I have added the following text describing the criteria "We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses. " As for what counts as “correct,” “appropriate,” and “incorrect”, I have the following text "We assessed the translation (in)correctness and appropriateness of dialectical terms between the two AI models, ChatGPT and DeepSeek, and human translation. We considered the translation of an item correct if cultural and linguistic aspects are both maintained in the translation of this term. However, if one of these aspects is violated in the translation we consider it appropriate, and if both the cultural and linguistic aspects are violate in the translation, we consider it incorrect. " Important details are currently missing for full replication. In particular, these aspects need to be clarified: Model versions & dates: Report exact model versions/variants, query timestamps (LLMs change over time), and any API/app settings. Prompts: Publish the exact prompts, instructions (e.g., “translate to English; provide multiple senses?”), temperature/decoding parameters, number of attempts/retries, and whether any post-processing was applied. Text normalization: Specify Unicode normalization, diacritic handling, tokenization, and whether Arabic script was normalized before translation. Response Thank you very much for your valuable comment. I have addressed these aspects, but for the word limit I couldn't cover them all, which need a full-fledged new paper. Temper general claims about “SA bias” and cross-dialect performance; frame as evidence from this 50-term sample and invite replication on larger, balanced sets. Highlight that performance may change with contextual sentences and few-shot prompting; consider adding a small follow-up experiment (even in supplementary) to show sensitivity to prompt design. Response Thank you very much for your valuable comment. I have addressed these aspects, please see section 7. Terminology consistency: Use “YSA” consistently (a few instances seem to vary). Typos & phrasing (examples): “concisderable” → considerable; “corp” → crops (re: silo); “wording”/“orthograpgical” → orthographical; ensure consistent capitalization of model names and sections. Response Thank you very much for your valuable comment. I have addressed these aspects, revising the paper carefully for these issues. Finally, thank you very much once again for your valuable comments. Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point below. Thank you for the opportunity to review this paper. I appreciate the effort that went into this research and the attempt to apply computational methods to analyze diplomatic discourse. The topic is timely and valuable for MT/LLM evaluation beyond standard Arabic. The manuscript is generally clear, and the dataset is shared. Response Thank you very much for your valuable remark. Mostly clear, with a good high-level MT background. Literature is broadly cited, but several claims (e.g., model capabilities, architecture details, and comparative statements) would benefit from primary/technical citations and tighter focus on dialect MT work. In particular, these points need improvement: Tighten the background to emphasize prior work on Arabic dialect MT and dialectal evaluation, and separate general MT history from directly relevant work. Add citations for specific DeepSeek details you reference (Mixture-of-Experts, training corpora) and for Arabic dialect evaluation benchmarks/tools where applicable. Copy-edit for typos/wording (examples in Minor comments) Response Thank you very much for your valuable comment. I have addressed these aspects, adding a subsection dubbed as "3.1. Translating Arabic dialects", and added several references as recommended by you and the other reviewer. The exploratory design (50 terms) is a reasonable pilot, but sampling and gold-standard construction need more rigor to support conclusions. However, these points need modification: Sampling: Clarify how the 50 terms were selected (criteria, sources, representativeness across categories; frequency in real usage). Consider expanding to include contextualized uses (sentences) alongside isolated terms to reduce ambiguity. Gold standard: Specify the annotation protocol (number of native speakers, expertise, independence, adjudication process). Report inter-annotator agreement (e.g., Cohen’s κ) if multiple raters were used or describe how disagreements were resolved. Task framing: Define what counts as “correct,” “appropriate,” and “incorrect” a priori , with examples. Consider an error taxonomy (literal, SA-bias, cultural connotation loss, POS confusion, etc.). Response Thank you very much for your valuable comment. I have addressed these aspects, detailing the criteria adopted, defining "what counts as “correct,” “appropriate,” and “incorrect”, and other related aspects. For example, I have added the following text describing the criteria "We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses. " As for what counts as “correct,” “appropriate,” and “incorrect”, I have the following text "We assessed the translation (in)correctness and appropriateness of dialectical terms between the two AI models, ChatGPT and DeepSeek, and human translation. We considered the translation of an item correct if cultural and linguistic aspects are both maintained in the translation of this term. However, if one of these aspects is violated in the translation we consider it appropriate, and if both the cultural and linguistic aspects are violate in the translation, we consider it incorrect. " Important details are currently missing for full replication. In particular, these aspects need to be clarified: Model versions & dates: Report exact model versions/variants, query timestamps (LLMs change over time), and any API/app settings. Prompts: Publish the exact prompts, instructions (e.g., “translate to English; provide multiple senses?”), temperature/decoding parameters, number of attempts/retries, and whether any post-processing was applied. Text normalization: Specify Unicode normalization, diacritic handling, tokenization, and whether Arabic script was normalized before translation. Response Thank you very much for your valuable comment. I have addressed these aspects, but for the word limit I couldn't cover them all, which need a full-fledged new paper. Temper general claims about “SA bias” and cross-dialect performance; frame as evidence from this 50-term sample and invite replication on larger, balanced sets. Highlight that performance may change with contextual sentences and few-shot prompting; consider adding a small follow-up experiment (even in supplementary) to show sensitivity to prompt design. Response Thank you very much for your valuable comment. I have addressed these aspects, please see section 7. Terminology consistency: Use “YSA” consistently (a few instances seem to vary). Typos & phrasing (examples): “concisderable” → considerable; “corp” → crops (re: silo); “wording”/“orthograpgical” → orthographical; ensure consistent capitalization of model names and sections. Response Thank you very much for your valuable comment. I have addressed these aspects, revising the paper carefully for these issues. Finally, thank you very much once again for your valuable comments. Competing Interests: No competing interests were disclosed. Close Report a concern Respond or Comment COMMENTS ON THIS REPORT Author Response 02 Jan 2026 Mohammed Q. Shormani , English Studies, Ibb University, Ibb, Yemen 02 Jan 2026 Author Response Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find ... Continue reading Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point below. Thank you for the opportunity to review this paper. I appreciate the effort that went into this research and the attempt to apply computational methods to analyze diplomatic discourse. The topic is timely and valuable for MT/LLM evaluation beyond standard Arabic. The manuscript is generally clear, and the dataset is shared. Response Thank you very much for your valuable remark. Mostly clear, with a good high-level MT background. Literature is broadly cited, but several claims (e.g., model capabilities, architecture details, and comparative statements) would benefit from primary/technical citations and tighter focus on dialect MT work. In particular, these points need improvement: Tighten the background to emphasize prior work on Arabic dialect MT and dialectal evaluation, and separate general MT history from directly relevant work. Add citations for specific DeepSeek details you reference (Mixture-of-Experts, training corpora) and for Arabic dialect evaluation benchmarks/tools where applicable. Copy-edit for typos/wording (examples in Minor comments) Response Thank you very much for your valuable comment. I have addressed these aspects, adding a subsection dubbed as "3.1. Translating Arabic dialects", and added several references as recommended by you and the other reviewer. The exploratory design (50 terms) is a reasonable pilot, but sampling and gold-standard construction need more rigor to support conclusions. However, these points need modification: Sampling: Clarify how the 50 terms were selected (criteria, sources, representativeness across categories; frequency in real usage). Consider expanding to include contextualized uses (sentences) alongside isolated terms to reduce ambiguity. Gold standard: Specify the annotation protocol (number of native speakers, expertise, independence, adjudication process). Report inter-annotator agreement (e.g., Cohen’s κ) if multiple raters were used or describe how disagreements were resolved. Task framing: Define what counts as “correct,” “appropriate,” and “incorrect” a priori , with examples. Consider an error taxonomy (literal, SA-bias, cultural connotation loss, POS confusion, etc.). Response Thank you very much for your valuable comment. I have addressed these aspects, detailing the criteria adopted, defining "what counts as “correct,” “appropriate,” and “incorrect”, and other related aspects. For example, I have added the following text describing the criteria "We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses. " As for what counts as “correct,” “appropriate,” and “incorrect”, I have the following text "We assessed the translation (in)correctness and appropriateness of dialectical terms between the two AI models, ChatGPT and DeepSeek, and human translation. We considered the translation of an item correct if cultural and linguistic aspects are both maintained in the translation of this term. However, if one of these aspects is violated in the translation we consider it appropriate, and if both the cultural and linguistic aspects are violate in the translation, we consider it incorrect. " Important details are currently missing for full replication. In particular, these aspects need to be clarified: Model versions & dates: Report exact model versions/variants, query timestamps (LLMs change over time), and any API/app settings. Prompts: Publish the exact prompts, instructions (e.g., “translate to English; provide multiple senses?”), temperature/decoding parameters, number of attempts/retries, and whether any post-processing was applied. Text normalization: Specify Unicode normalization, diacritic handling, tokenization, and whether Arabic script was normalized before translation. Response Thank you very much for your valuable comment. I have addressed these aspects, but for the word limit I couldn't cover them all, which need a full-fledged new paper. Temper general claims about “SA bias” and cross-dialect performance; frame as evidence from this 50-term sample and invite replication on larger, balanced sets. Highlight that performance may change with contextual sentences and few-shot prompting; consider adding a small follow-up experiment (even in supplementary) to show sensitivity to prompt design. Response Thank you very much for your valuable comment. I have addressed these aspects, please see section 7. Terminology consistency: Use “YSA” consistently (a few instances seem to vary). Typos & phrasing (examples): “concisderable” → considerable; “corp” → crops (re: silo); “wording”/“orthograpgical” → orthographical; ensure consistent capitalization of model names and sections. Response Thank you very much for your valuable comment. I have addressed these aspects, revising the paper carefully for these issues. Finally, thank you very much once again for your valuable comments. Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point below. Thank you for the opportunity to review this paper. I appreciate the effort that went into this research and the attempt to apply computational methods to analyze diplomatic discourse. The topic is timely and valuable for MT/LLM evaluation beyond standard Arabic. The manuscript is generally clear, and the dataset is shared. Response Thank you very much for your valuable remark. Mostly clear, with a good high-level MT background. Literature is broadly cited, but several claims (e.g., model capabilities, architecture details, and comparative statements) would benefit from primary/technical citations and tighter focus on dialect MT work. In particular, these points need improvement: Tighten the background to emphasize prior work on Arabic dialect MT and dialectal evaluation, and separate general MT history from directly relevant work. Add citations for specific DeepSeek details you reference (Mixture-of-Experts, training corpora) and for Arabic dialect evaluation benchmarks/tools where applicable. Copy-edit for typos/wording (examples in Minor comments) Response Thank you very much for your valuable comment. I have addressed these aspects, adding a subsection dubbed as "3.1. Translating Arabic dialects", and added several references as recommended by you and the other reviewer. The exploratory design (50 terms) is a reasonable pilot, but sampling and gold-standard construction need more rigor to support conclusions. However, these points need modification: Sampling: Clarify how the 50 terms were selected (criteria, sources, representativeness across categories; frequency in real usage). Consider expanding to include contextualized uses (sentences) alongside isolated terms to reduce ambiguity. Gold standard: Specify the annotation protocol (number of native speakers, expertise, independence, adjudication process). Report inter-annotator agreement (e.g., Cohen’s κ) if multiple raters were used or describe how disagreements were resolved. Task framing: Define what counts as “correct,” “appropriate,” and “incorrect” a priori , with examples. Consider an error taxonomy (literal, SA-bias, cultural connotation loss, POS confusion, etc.). Response Thank you very much for your valuable comment. I have addressed these aspects, detailing the criteria adopted, defining "what counts as “correct,” “appropriate,” and “incorrect”, and other related aspects. For example, I have added the following text describing the criteria "We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses. " As for what counts as “correct,” “appropriate,” and “incorrect”, I have the following text "We assessed the translation (in)correctness and appropriateness of dialectical terms between the two AI models, ChatGPT and DeepSeek, and human translation. We considered the translation of an item correct if cultural and linguistic aspects are both maintained in the translation of this term. However, if one of these aspects is violated in the translation we consider it appropriate, and if both the cultural and linguistic aspects are violate in the translation, we consider it incorrect. " Important details are currently missing for full replication. In particular, these aspects need to be clarified: Model versions & dates: Report exact model versions/variants, query timestamps (LLMs change over time), and any API/app settings. Prompts: Publish the exact prompts, instructions (e.g., “translate to English; provide multiple senses?”), temperature/decoding parameters, number of attempts/retries, and whether any post-processing was applied. Text normalization: Specify Unicode normalization, diacritic handling, tokenization, and whether Arabic script was normalized before translation. Response Thank you very much for your valuable comment. I have addressed these aspects, but for the word limit I couldn't cover them all, which need a full-fledged new paper. Temper general claims about “SA bias” and cross-dialect performance; frame as evidence from this 50-term sample and invite replication on larger, balanced sets. Highlight that performance may change with contextual sentences and few-shot prompting; consider adding a small follow-up experiment (even in supplementary) to show sensitivity to prompt design. Response Thank you very much for your valuable comment. I have addressed these aspects, please see section 7. Terminology consistency: Use “YSA” consistently (a few instances seem to vary). Typos & phrasing (examples): “concisderable” → considerable; “corp” → crops (re: silo); “wording”/“orthograpgical” → orthographical; ensure consistent capitalization of model names and sections. Response Thank you very much for your valuable comment. I have addressed these aspects, revising the paper carefully for these issues. Finally, thank you very much once again for your valuable comments. Competing Interests: No competing interests were disclosed. Close Report a concern COMMENT ON THIS REPORT Comments on this article Comments (0) Version 2 VERSION 2 PUBLISHED 16 Jul 2025 ADD YOUR COMMENT Comment keyboard_arrow_left keyboard_arrow_right Open Peer Review Reviewer Status info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Reviewer Reports Invited Reviewers 1 2 3 Version 2 (revision) 16 Jan 26 read read Version 1 16 Jul 25 read read Cao-Tuong DINH , FPT University, Can Tho city, Vietnam Liang Ding , The University of Sydney, Sydney, Australia Hend S Al-Khalifa , King Saud University, Riyadh, Saudi Arabia Comments on this article All Comments (0) Add a comment Sign up for content alerts Sign Up You are now signed up to receive this alert Browse by related subjects keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2026 Al-Khalifa H. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 06 Mar 2026 | for Version 2 Hend S Al-Khalifa , King Saud University, Riyadh, Saudi Arabia 0 Views copyright © 2026 Al-Khalifa H. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Not Approved info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Here is the refined review rewritten in coherent paragraph form: The introduction should clearly and explicitly state the research objectives and research questions at the beginning of the manuscript, rather than introducing them later in the paper. Presenting these elements upfront would help readers better understand the scope and motivation of the study from the outset. Section 2 contains only one subsection (2.1), making this subdivision unnecessary in the absence of additional subsections. Furthermore, Section 2.1 is overly verbose and includes repeated information. Much of this content could be streamlined and more effectively presented using figures and tables. Similarly, Section 3 also contains only one subsection (3.1), which is unnecessary given the lack of further subdivisions. In addition, although this section is titled “Translating Dialects,” its content primarily focuses on translation between languages rather than dialectal translation. This mismatch reduces the clarity and coherence of the section. It would be more appropriate to include a dedicated section that specifically addresses interlanguage translation. The exclusive use of ChatGPT and DeepSeek in the study is not sufficiently justified. Evaluating a broader range of large language models would strengthen the empirical foundation of the work and enhance its contribution. Moreover, the dataset used in the study is very limited in size and does not appear to support a substantial scientific contribution. The criteria for selecting the categories and lexical items are also not clearly explained, which further weakens the methodological rigor. The methodological procedure lacks sufficient detail and transparency. In particular, the prompts used for generating translations are not described. It remains unclear whether zero-shot or few-shot prompting strategies were employed and whether the prompts were formulated in English or Arabic. In addition, the evaluation process is insufficiently documented. The number of human evaluators is not reported, and no information is provided regarding inter-annotator agreement, making it difficult to assess the reliability of the results. The discussion section largely reiterates the findings already presented in the Results section, rather than providing deeper interpretation, critical analysis, or theoretical engagement. Similarly, the section on “Linguistic Subtleties” is superficial and would benefit from a more rigorous engagement with established linguistic theories and relevant scholarly literature. The conclusion of the paper is largely generic and reflects observations that could apply to most studies on low-resource or dialectal Arabic. As a result, it does not sufficiently highlight the specific contributions or implications of the present study. Overall, the manuscript is excessively verbose, with repeated details appearing across multiple sections and subsections that often stand alone. This structural imbalance negatively affects the coherence, readability, and flow of the paper. Greater concision and more careful organization would substantially improve the quality and impact of the manuscript. Is the work clearly and accurately presented and does it cite the current literature? Partly Is the study design appropriate and is the work technically sound? No Are sufficient details of methods and analysis provided to allow replication by others? No If applicable, is the statistical analysis and its interpretation appropriate? Not applicable Are all the source data underlying the results available to ensure full reproducibility? Partly Are the conclusions drawn adequately supported by the results? Partly Competing Interests No competing interests were disclosed. Reviewer Expertise Arabic NLP I confirm that I have read this submission and believe that I have an appropriate level of expertise to state that I do not consider it to be of an acceptable scientific standard, for reasons outlined above. reply Respond to this report Responses (0) Al-Khalifa HS. Peer Review Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.194488.r460322) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/14-694/v2#referee-response-460322 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2026 Ding L. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 20 Jan 2026 | for Version 2 Liang Ding , The University of Sydney, Sydney, Australia 0 Views copyright © 2026 Ding L. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Thank the authors for addressing most of my concerns. Competing Interests No competing interests were disclosed. Reviewer Expertise natural language processing, machine learning, large language models, machine translation I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. reply Respond to this report Responses (0) Ding L. Peer Review Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.194488.r450599) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/14-694/v2#referee-response-450599 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2025 Ding L. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 25 Sep 2025 | for Version 1 Liang Ding , The University of Sydney, Sydney, Australia 0 Views copyright © 2025 Ding L. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (1) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions This study evaluates the performance of two large language models, ChatGPT-4o and DeepSeek v3, on the task of translating 50 culturally specific dialectical terms from Yemeni Sana'ani Arabic (YSA) into English. The authors perform a qualitative and quantitative analysis, comparing the models' outputs against a human-provided translation. The core findings indicate that both models struggle significantly with the nuances of the YSA dialect, frequently defaulting to Standard Arabic (SA) interpretations or producing literal, contextually incorrect translations. The study concludes that ChatGPT performs marginally better than DeepSeek, but both are inadequate for reliable dialect translation, highlighting a critical need for dialect-aware data and model development. Pros: 1) The translation of low-resource and non-standard dialects is a critical frontier for machine translation. This work addresses a significant gap by focusing on YSA, an understudied variety, providing valuable initial insights into the limitations of current state-of-the-art LLMs. 2) To my knowledge, this is one of the first academic papers to benchmark the DeepSeek model against ChatGPT on a dialectal translation task, making a timely contribution to the comparative analysis of emerging LLMs. 3) The paper excels in its qualitative discussion. The term-by-term breakdown in Table 2 and the subsequent discussion of cultural, linguistic, and contextual subtleties (Sections 5.2 and 6) provide clear, compelling evidence of the models' failures and the reasons behind them (e.g., mistaking شركة for 'company' instead of 'meat'). Cons: The paper, while valuable for its topic, is constrained by significant experimental and contextual limitations that temper its conclusions. 1) The analysis relies solely on the authors' classification of translations as "correct," "incorrect," or "appropriate". This is highly subjective and lacks the rigor of standard MT evaluation. State-of-the-art research has moved beyond lexical-overlap metrics like BLEU to model-based metrics such as COMET, which better capture semantic fidelity. The absence of such metrics makes the quantitative claims (e.g., 34% correct for ChatGPT) difficult to verify and compare against other work. 2) The study's methodology does not reflect the current best practices for evaluating LLMs in translation tasks. No Prompt Engineering: The paper fails to specify the prompts used, but the results suggest a simple, zero-shot directive (e.g., "Translate X"). This is a critical flaw. It is well-documented that LLM performance is highly sensitive to prompting strategies. The models were not given a fair chance to perform well. Term-Level Translation: The evaluation is conducted on isolated terms. This is not representative of real-world usage and ignores the crucial role of context in disambiguation. Translating the terms within full sentences would have provided a more realistic and challenging test. 3) The recommendations for AI developers are generic (e.g., "expand the dialectal data coverage"). The paper would be substantially strengthened by engaging with more specific, high-impact research in machine translation to offer more sophisticated solutions. To address these shortcomings, the authors should incorporate and cite the following highly relevant papers : Peng, K. et al. (2023) [1]. Towards Making the Most of ChatGPT for Machine Translation : This paper is essential reading. It explicitly demonstrates that simple prompting limits ChatGPT's translation ability. The authors should have experimented with the Task-Specific Prompts (TSP) and Domain-Specific Prompts (DSP) are proposed in this work to see if defining the task more clearly ("You are a machine translation system translating the Yemeni Sana'ani dialect") improves performance. Furthermore, this paper advocates for using COMET as a primary metric, a practice this study should adopt. Zan, C. et al. (2022) [2]. Vega-MT: The JD Explore Academy Translation System for WMT22 : While focused on a traditional NMT system, this paper offers a blueprint for building high-quality multilingual models. Its concept of "multi-directional pretraining" —using data from all language pairs to exploit common knowledge—is a concrete technical strategy that goes beyond the paper's generic recommendation to simply "add more data". The authors could propose a "multi-dialectal" pre-training approach inspired by this work as a specific path forward. Ding, L. et al. (2021) [3]. Improving Neural Machine Translation by Bidirectional Training: This paper introduces Bidirectional Training (BiT) , a simple yet effective strategy of pre-training a model on both src -> tgt and tgt -> src data simultaneously. This method was shown to improve performance, especially in low-resource settings. Suggesting a BiT-based fine-tuning approach (using both YSA-to-English and English-to-YSA data) would be a specific, testable hypothesis for improving the models' grasp of YSA's linguistic structure and improving bilingual alignment. Ref: [1] Peng et al. Towards Making the Most of ChatGPT for Machine Translation. In Findings of EMNLP 2023. [2] Zan et al., Vega-MT: The JD Explore Academy Translation System for WMT22. In WMT 2022. [3] Ding et al., Improving Neural Machine Translation by Bidirectional Training. In EMNLP 2021. Is the work clearly and accurately presented and does it cite the current literature? Yes Is the study design appropriate and is the work technically sound? Yes Are sufficient details of methods and analysis provided to allow replication by others? Yes If applicable, is the statistical analysis and its interpretation appropriate? Yes Are all the source data underlying the results available to ensure full reproducibility? No source data required Are the conclusions drawn adequately supported by the results? Yes Competing Interests No competing interests were disclosed. Reviewer Expertise natural language processing, machine learning, large language models, machine translation I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (1) Author Response 02 Jan 2026 Mohammed Q. Shormani, English Studies, Ibb University, Ibb, Yemen Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point as follows. This study evaluates the performance of two large language models, ChatGPT-4o and DeepSeek v3, on the task of translating 50 culturally specific dialectical terms from Yemeni Sana'ani Arabic (YSA) into English. The authors perform a qualitative and quantitative analysis, comparing the models' outputs against a human-provided translation. The core findings indicate that both models struggle significantly with the nuances of the YSA dialect, frequently defaulting to Standard Arabic (SA) interpretations or producing literal, contextually incorrect translations. The study concludes that ChatGPT performs marginally better than DeepSeek, but both are inadequate for reliable dialect translation, highlighting a critical need for dialect-aware data and model development. Response Thank you very much for your valuable remark. 1)The translation of low-resource and non-standard dialects is a critical frontier for machine translation. This work addresses a significant gap by focusing on YSA, an understudied variety, providing valuable initial insights into the limitations of current state-of-the-art LLMs. Response Thank you very much for your valuable remark. 2) To my knowledge, this is one of the first academic papers to benchmark the DeepSeek model against ChatGPT on a dialectal translation task, making a timely contribution to the comparative analysis of emerging LLMs. Response Thank you very much for your valuable remark. 3) The paper excels in its qualitative discussion. The term-by-term breakdown in Table 2 and the subsequent discussion of cultural, linguistic, and contextual subtleties (Sections 5.2 and 6) provide clear, compelling evidence of the models' failures and the reasons behind them (e.g., mistaking شركة for 'company' instead of 'meat'). Response Thank you very much for your valuable remark. The analysis relies solely on the authors' classification of translations as "correct," "incorrect," or "appropriate". This is highly subjective and lacks the rigor of standard MT evaluation. State-of-the-art research has moved beyond lexical-overlap metrics like BLEU to model-based metrics such as COMET, which better capture semantic fidelity. The absence of such metrics makes the quantitative claims (e.g., 34% correct for ChatGPT) difficult to verify and compare against other work. Response Thank you very much for your valuable comment. Please note that our use of the categories correct, incorrect, and appropriate was intentionally motivated by the nature of the task: translating Sana’ni Arabic, a dialect for which no standardized reference corpora or validated automatic evaluation benchmarks currently exist. As a result, commonly used model-based metrics such as COMET—which require high-quality reference translations and have been trained predominantly on Standard Arabic and high-resource language pairs—are not yet reliable indicators of translation quality for this dialect. I would like also to clarify that our quantitative results as indicative rather than absolute. I have also developed this part by incoporating the following text " We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses." 2) The study's methodology does not reflect the current best practices for evaluating LLMs in translation tasks. No Prompt Engineering: The paper fails to specify the prompts used, but the results suggest a simple, zero-shot directive (e.g., "Translate X"). This is a critical flaw. It is well-documented that LLM performance is highly sensitive to prompting strategies. The models were not given a fair chance to perform well. Term-Level Translation: The evaluation is conducted on isolated terms. This is not representative of real-world usage and ignores the crucial role of context in disambiguation. Translating the terms within full sentences would have provided a more realistic and challenging test. Response Thank you very much for your valuable comment. I have addressed this aspect, and acknowledged it as one limitation of the study. Please section 7. The recommendations for AI developers are generic (e.g., "expand the dialectal data coverage"). The paper would be substantially strengthened by engaging with more specific, high-impact research in machine translation to offer more sophisticated solutions....... Response Thank you very much for your valuable comment. I have addressed these outstanding points and referred to these references. Please see section 7 [1] Peng et al. Towards Making the Most of ChatGPT for Machine Translation. In Findings of EMNLP 2023. [2] Zan et al., Vega-MT: The JD Explore Academy Translation System for WMT22. In WMT 2022. [3] Ding et al., Improving Neural Machine Translation by Bidirectional Training. In EMNLP 2021. Finally, thank you very much for your valuable comments and for the efforts you exerted to improving the article/ View more View less Competing Interests No competing interest to disclose. reply Respond Report a concern Ding L. Peer Review Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.182657.r403515) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/14-694/v1#referee-response-403515 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2025 DINH C. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. The author(s) is/are employees of the US Government and therefore domestic copyright protection in USA does not apply to this work. The work may be protected under the copyright laws of other jurisdictions when used in those jurisdictions. 24 Sep 2025 | for Version 1 Cao-Tuong DINH , FPT University, Can Tho city, Vietnam 0 Views copyright © 2025 DINH C. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. The author(s) is/are employees of the US Government and therefore domestic copyright protection in USA does not apply to this work. The work may be protected under the copyright laws of other jurisdictions when used in those jurisdictions. format_quote Cite this report speaker_notes Responses (1) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Recommendation: Major revisions Required For author and editor Thank you for the opportunity to review this paper. I appreciate the effort that went into this research and the attempt to apply computational methods to analyze diplomatic discourse. The topic is timely and valuable for MT/LLM evaluation beyond standard Arabic. The manuscript is generally clear, and the dataset is shared. However, methodological transparency and rigor need strengthening (sampling, prompting, validation, and statistics). A few internal inconsistencies and copy-editing issues also require attention. With moderate revisions, the paper will provide a useful exploratory baseline, hence Approved with major revision . However, as a researcher, I have some concerns as below: 1) Clarity & literature context Mostly clear, with a good high-level MT background. Literature is broadly cited, but several claims (e.g., model capabilities, architecture details, and comparative statements) would benefit from primary/technical citations and tighter focus on dialect MT work. In particular, these points need improvement: Tighten the background to emphasize prior work on Arabic dialect MT and dialectal evaluation, and separate general MT history from directly relevant work. Add citations for specific DeepSeek details you reference (Mixture-of-Experts, training corpora) and for Arabic dialect evaluation benchmarks/tools where applicable. Copy-edit for typos/wording (examples in Minor comments). 2) Study design & academic merit: The exploratory design (50 terms) is a reasonable pilot, but sampling and gold-standard construction need more rigor to support conclusions. However, these points need modification: Sampling: Clarify how the 50 terms were selected (criteria, sources, representativeness across categories; frequency in real usage). Consider expanding to include contextualized uses (sentences) alongside isolated terms to reduce ambiguity. Gold standard: Specify the annotation protocol (number of native speakers, expertise, independence, adjudication process). Report inter-annotator agreement (e.g., Cohen’s κ) if multiple raters were used or describe how disagreements were resolved. Task framing: Define what counts as “correct,” “appropriate,” and “incorrect” a priori , with examples. Consider an error taxonomy (literal, SA-bias, cultural connotation loss, POS confusion, etc.). 3) Methods detail & reproducibility: Important details are currently missing for full replication. In particular, these aspects need to be clarified: Model versions & dates: Report exact model versions/variants, query timestamps (LLMs change over time), and any API/app settings. Prompts: Publish the exact prompts, instructions (e.g., “translate to English; provide multiple senses?”), temperature/decoding parameters, number of attempts/retries, and whether any post-processing was applied. Text normalization: Specify Unicode normalization, diacritic handling, tokenization, and whether Arabic script was normalized before translation. 4) Statistics & interpretation: Counts and percentages are reported, but statistical comparison is minimal. Below are suggestions for improvement: Add 95% CIs for accuracy/“correct” rates by category and overall. For paired categorical outcomes (ChatGPT vs DeepSeek on the same items), apply a McNemar test (or exact variant) to test whether differences are statistically significant. Report per-category performance with CIs and consider a simple mixed-effects model (random intercept for term) to account for item variability. If you keep “appropriate” as a middle category, consider ordinal models or report separate binary analyses (correct vs not; correct+appropriate vs incorrect). 5) Conclusions vs results: Broadly aligned, but some statements verge on over-generalization given sample size and item selection. Below are suggestions for improvement: Temper general claims about “SA bias” and cross-dialect performance; frame as evidence from this 50-term sample and invite replication on larger, balanced sets. Highlight that performance may change with contextual sentences and few-shot prompting; consider adding a small follow-up experiment (even in supplementary) to show sensitivity to prompt design. 6) Minor comments (clarity, style, presentation) Terminology consistency: Use “YSA” consistently (a few instances seem to vary). Typos & phrasing (examples): “concisderable” → considerable; “corp” → crops (re: silo); “wording”/“orthograpgical” → orthographical; ensure consistent capitalization of model names and sections. Is the work clearly and accurately presented and does it cite the current literature? Partly Is the study design appropriate and is the work technically sound? Partly Are sufficient details of methods and analysis provided to allow replication by others? Partly If applicable, is the statistical analysis and its interpretation appropriate? Partly Are all the source data underlying the results available to ensure full reproducibility? Partly Are the conclusions drawn adequately supported by the results? Partly Competing Interests No competing interests were disclosed. Reviewer Expertise self-regulated learning, EMI, technology-based teaching & learning in higher education I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (1) Author Response 02 Jan 2026 Mohammed Q. Shormani, English Studies, Ibb University, Ibb, Yemen Dear reviewer, First, let me express my sincere appreciation and thanks for the insightful comments you have provided, which have improved the paper in form and content. Please find our responses to your comments point-by-point below. Thank you for the opportunity to review this paper. I appreciate the effort that went into this research and the attempt to apply computational methods to analyze diplomatic discourse. The topic is timely and valuable for MT/LLM evaluation beyond standard Arabic. The manuscript is generally clear, and the dataset is shared. Response Thank you very much for your valuable remark. Mostly clear, with a good high-level MT background. Literature is broadly cited, but several claims (e.g., model capabilities, architecture details, and comparative statements) would benefit from primary/technical citations and tighter focus on dialect MT work. In particular, these points need improvement: Tighten the background to emphasize prior work on Arabic dialect MT and dialectal evaluation, and separate general MT history from directly relevant work. Add citations for specific DeepSeek details you reference (Mixture-of-Experts, training corpora) and for Arabic dialect evaluation benchmarks/tools where applicable. Copy-edit for typos/wording (examples in Minor comments) Response Thank you very much for your valuable comment. I have addressed these aspects, adding a subsection dubbed as "3.1. Translating Arabic dialects", and added several references as recommended by you and the other reviewer. The exploratory design (50 terms) is a reasonable pilot, but sampling and gold-standard construction need more rigor to support conclusions. However, these points need modification: Sampling: Clarify how the 50 terms were selected (criteria, sources, representativeness across categories; frequency in real usage). Consider expanding to include contextualized uses (sentences) alongside isolated terms to reduce ambiguity. Gold standard: Specify the annotation protocol (number of native speakers, expertise, independence, adjudication process). Report inter-annotator agreement (e.g., Cohen’s κ) if multiple raters were used or describe how disagreements were resolved. Task framing: Define what counts as “correct,” “appropriate,” and “incorrect” a priori , with examples. Consider an error taxonomy (literal, SA-bias, cultural connotation loss, POS confusion, etc.). Response Thank you very much for your valuable comment. I have addressed these aspects, detailing the criteria adopted, defining "what counts as “correct,” “appropriate,” and “incorrect”, and other related aspects. For example, I have added the following text describing the criteria "We collected the study data from different sources on the web, such as Wikipedia and the first author’s relatives, who are native speakers of Sana’ni Arabic. The dataset contained 50 terms. Our criteria of selecting these 50 terms include: i) they should be representative, viz., belonging to places, clothes, animals, stuff, household s., ii) they should give us enough room to have a representative sample on the aspects of YSA culture, iii) they should represent all lexical items, viz., nouns, verbs, adjectives, and adverbs, and iv) they should allow us both linguistic and cultural analyses. " As for what counts as “correct,” “appropriate,” and “incorrect”, I have the following text "We assessed the translation (in)correctness and appropriateness of dialectical terms between the two AI models, ChatGPT and DeepSeek, and human translation. We considered the translation of an item correct if cultural and linguistic aspects are both maintained in the translation of this term. However, if one of these aspects is violated in the translation we consider it appropriate, and if both the cultural and linguistic aspects are violate in the translation, we consider it incorrect. " Important details are currently missing for full replication. In particular, these aspects need to be clarified: Model versions & dates: Report exact model versions/variants, query timestamps (LLMs change over time), and any API/app settings. Prompts: Publish the exact prompts, instructions (e.g., “translate to English; provide multiple senses?”), temperature/decoding parameters, number of attempts/retries, and whether any post-processing was applied. Text normalization: Specify Unicode normalization, diacritic handling, tokenization, and whether Arabic script was normalized before translation. Response Thank you very much for your valuable comment. I have addressed these aspects, but for the word limit I couldn't cover them all, which need a full-fledged new paper. Temper general claims about “SA bias” and cross-dialect performance; frame as evidence from this 50-term sample and invite replication on larger, balanced sets. Highlight that performance may change with contextual sentences and few-shot prompting; consider adding a small follow-up experiment (even in supplementary) to show sensitivity to prompt design. Response Thank you very much for your valuable comment. I have addressed these aspects, please see section 7. Terminology consistency: Use “YSA” consistently (a few instances seem to vary). Typos & phrasing (examples): “concisderable” → considerable; “corp” → crops (re: silo); “wording”/“orthograpgical” → orthographical; ensure consistent capitalization of model names and sections. Response Thank you very much for your valuable comment. I have addressed these aspects, revising the paper carefully for these issues. Finally, thank you very much once again for your valuable comments. View more View less Competing Interests No competing interests were disclosed. reply Respond Report a concern DINH CT. Peer Review Report For: Translating dialects between ChatGPT and DeepSeek: Yemeni San’ani Arabic terms as a case-in-point [version 2; peer review: 1 approved, 1 approved with reservations, 1 not approved] . F1000Research 2026, 14 :694 ( https://doi.org/10.5256/f1000research.182657.r409822) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/14-694/v1#referee-response-409822 Alongside their report, reviewers assign a status to the article: Approved - the paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations - A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved - fundamental flaws in the paper seriously undermine the findings and conclusions Adjust parameters to alter display View on desktop for interactive features Includes Interactive Elements View on desktop for interactive features Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Stay Updated Sign up for content alerts and receive a weekly or monthly email with all newly published articles Register with F1000Research Already registered? Sign in Not now, thanks close PLEASE NOTE If you are an AUTHOR of this article, please check that you signed in with the account associated with this article otherwise we cannot automatically identify your role as an author and your comment will be labelled as a “User Comment”. If you are a REVIEWER of this article, please check that you have signed in with the account associated with this article and then go to your account to submit your report, please do not post your review here. If you do not have access to your original account, please contact us . All commenters must hold a formal affiliation as per our Policies . The information that you give us will be displayed next to your comment. User comments must be in English, comprehensible and relevant to the article under discussion. We reserve the right to remove any comments that we consider to be inappropriate, offensive or otherwise in breach of the User Comment Terms and Conditions . Commenters must not use a comment for personal attacks. When criticisms of the article are based on unpublished data, the data should be made available. I accept the User Comment Terms and Conditions Please confirm that you accept the User Comment Terms and Conditions. Affiliation ✕ refresh Please enter your institution. Note: To add your institution or organisation, start typing the name and then select the correct name from the list. Where applicable, the name will appear in both the original language and in English. Do not paste in the name. If the name does not appear in the drop-down list, we will display the information you have entered. ✕ refresh Country/Region * USA UK Canada China France Germany Afghanistan Aland Islands Albania Algeria American Samoa Andorra Angola Anguilla Antarctica Antigua and Barbuda Argentina Armenia Aruba Australia Austria Azerbaijan Bahamas Bahrain Bangladesh Barbados Belarus Belgium Belize Benin Bermuda Bhutan Bolivia Bosnia and Herzegovina Botswana Bouvet Island Brazil British Indian Ocean Territory British Virgin Islands Brunei Bulgaria Burkina Faso Burundi Cambodia Cameroon Canada Cape Verde Cayman Islands Central African Republic Chad Chile China Christmas Island Cocos (Keeling) Islands Colombia Comoros Congo Cook Islands Costa Rica Cote d'Ivoire Croatia Cuba Cyprus Czech Republic Democratic Republic of the Congo Denmark Djibouti Dominica Dominican Republic Ecuador Egypt El Salvador Equatorial Guinea Eritrea Estonia Ethiopia Falkland Islands Faroe Islands Federated States of Micronesia Fiji Finland France French Guiana French Polynesia French Southern Territories Gabon Georgia Germany Ghana Gibraltar Greece Greenland Grenada Guadeloupe Guam Guatemala Guernsey Guinea Guinea-Bissau Guyana Haiti Heard Island and Mcdonald Islands Holy See (Vatican City State) Honduras Hong Kong Hungary Iceland India Indonesia Iran Iraq Ireland Israel Italy Jamaica Japan Jersey Jordan Kazakhstan Kenya Kiribati Kosovo (Serbia and Montenegro) Kuwait Kyrgyzstan Lao People's Democratic Republic Latvia Lebanon Lesotho Liberia Libya Liechtenstein Lithuania Luxembourg Macao Madagascar Malawi Malaysia Maldives Mali Malta Marshall Islands Martinique Mauritania Mauritius Mayotte Mexico Minor Outlying Islands of the United States Moldova Monaco Mongolia Montenegro Montserrat Morocco Mozambique Myanmar Namibia Nauru Nepal Netherlands Antilles New Caledonia New Zealand Nicaragua Niger Nigeria Niue Norfolk Island North Korea North Macedonia Northern Mariana Islands Norway Oman Pakistan Palau Palestinian Territory Panama Papua New Guinea Paraguay Peru Philippines Pitcairn Poland Portugal Puerto Rico Qatar Reunion Romania Russian Federation Rwanda Saint Helena Saint Kitts and Nevis Saint Lucia Saint Pierre and Miquelon Saint Vincent and the Grenadines Samoa San Marino Sao Tome and Principe Saudi Arabia Senegal Serbia Seychelles Sierra Leone Singapore Slovakia Slovenia Solomon Islands Somalia South Africa South Georgia and the South Sandwich Is South Korea South Sudan Spain Sri Lanka Sudan Suriname Svalbard and Jan Mayen Swaziland Sweden Switzerland Syria Taiwan Tajikistan Tanzania Thailand The Gambia The Netherlands Timor-Leste Togo Tokelau Tonga Trinidad and Tobago Tunisia Turkey Turkmenistan Turks and Caicos Islands Tuvalu UK USA Uganda Ukraine United Arab Emirates United States Virgin Islands Uruguay Uzbekistan Vanuatu Venezuela Vietnam Wallis and Futuna West Bank and Gaza Strip Western Sahara Yemen Zambia Zimbabwe Please select your country/region. You must enter a comment. Competing Interests Please disclose any competing interests that might be construed to influence your judgment of the article's or peer review report's validity or importance. Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Please state your competing interests The comment has been saved. An error has occurred. Please try again. Cancel Post var lTitle = "Translating dialects between ChatGPT and...".replace("'", ''); var linkedInUrl = "http://www.linkedin.com/shareArticle?url=https://f1000research.com/articles/14-694/v2" + "&title=" + encodeURIComponent(lTitle) + "&summary=" + encodeURIComponent('Read the article by '); var deliciousUrl = "https://del.icio.us/post?url=https://f1000research.com/articles/14-694/v2&title=" + encodeURIComponent(lTitle); var redditUrl = "http://reddit.com/submit?url=https://f1000research.com/articles/14-694/v2" + "&title=" + encodeURIComponent(lTitle); linkedInUrl += encodeURIComponent('Shormani MQ and Al-Samki AA'); var offsetTop = /chrome/i.test( navigator.userAgent ) ? 4 : -10; var addthis_config = { ui_offset_top: offsetTop, services_compact : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_expanded : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_custom : [ { name: "LinkedIn", url: linkedInUrl, icon:"/img/icon/at_linkedin.svg" }, { name: "Mendeley", url: "http://www.mendeley.com/import/?url=https://f1000research.com/articles/14-694/v2/mendeley", icon:"/img/icon/at_mendeley.svg" }, { name: "Reddit", url: redditUrl, icon:"/img/icon/at_reddit.svg" }, ] }; var addthis_share = { url: "https://f1000research.com/articles/14-694", templates : { twitter : "Translating dialects between ChatGPT and DeepSeek: Yemeni San\’ani.... Shormani MQ and Al-Samki AA, published by " + "@F1000Research" + ", https://f1000research.com/articles/14-694/v2" } }; if (typeof(addthis) != "undefined"){ addthis.addEventListener('addthis.ready', checkCount); addthis.addEventListener('addthis.menu.share', checkCount); } $(".f1r-shares-twitter").attr("href", "https://twitter.com/intent/tweet?text=" + addthis_share.templates.twitter); $(".f1r-shares-facebook").attr("href", "https://www.facebook.com/sharer/sharer.php?u=" + addthis_share.url); $(".f1r-shares-linkedin").attr("href", addthis_config.services_custom[0].url); $(".f1r-shares-reddit").attr("href", addthis_config.services_custom[2].url); $(".f1r-shares-mendelay").attr("href", addthis_config.services_custom[1].url); function checkCount(){ setTimeout(function(){ $(".addthis_button_expanded").each(function(){ var count = $(this).text(); if (count !== "" && count != "0") $(this).removeClass("is-hidden"); else $(this).addClass("is-hidden"); }); }, 1000); } close How to cite this report {{reportCitation}} Cancel Copy Citation Details $(function(){R.ui.buttonDropdowns('.dropdown-for-downloads');}); $(function(){R.ui.toolbarDropdowns('.toolbar-dropdown-for-downloads');}); $.get("/articles/acj/165879/194488") new F1000.Clipboard(); new F1000.ThesaurusTermsDisplay("articles", "article", "194488"); $(document).ready(function() { $( "#frame1" ).on('load', function() { var mydiv = $(this).contents().find("div"); var h = mydiv.height(); console.log(h) }); var tooltipLivingFigure = jQuery(".interactive-living-figure-label .icon-more-info"), titleLivingFigure = tooltipLivingFigure.attr("title"); tooltipLivingFigure.simpletip({ fixed: true, position: ["-115", "30"], baseClass: 'small-tooltip', content:titleLivingFigure + " " }); tooltipLivingFigure.removeAttr("title"); $("body").on("click", ".cite-living-figure", function(e) { e.preventDefault(); var ref = $(this).attr("data-ref"); $(this).closest(".living-figure-list-container").find("#" + ref).fadeIn(200); }); $("body").on("click", ".close-cite-living-figure", function(e) { e.preventDefault(); $(this).closest(".popup-window-wrapper").fadeOut(200); }); $(document).on("mouseup", function(e) { var metricsContainer = $(".article-metrics-popover-wrapper"); if (!metricsContainer.is(e.target) && metricsContainer.has(e.target).length === 0) { $(".article-metrics-close-button").click(); } }); var articleId = $('#articleId').val(); if($("#main-article-count-box").attachArticleMetrics) { $("#main-article-count-box").attachArticleMetrics(articleId, { articleMetricsView: true }); } }); var figshareWidget = $(".new_figshare_widget"); if (figshareWidget.length > 0) { window.figshare.load("f1000", function(Widget) { // Select a tag/tags defined in your page. In this tag we will place the widget. _.map(figshareWidget, function(el){ var widget = new Widget({ articleId: $(el).attr("figshare_articleId") //height:300 // this is the height of the viewer part. [Default: 550] }); widget.initialize(); // initialize the widget widget.mount(el); // mount it in a tag that's on your page // this will save the widget on the global scope for later use from // your JS scripts. This line is optional. //window.widget = widget; }); }); } close Error Close Add Reset F1000.MICROSERVICES.AFFILIATION = ''; $(document).ready(function () { $('.js-affiliations-form').each((index, form) => { new AffiliationForm({ formId: form.id, institutionErrorSelector: '.comment-enter-institution', departmentErrorSelector: '.comment-enter-department', placeSelector: '.js-add-comment-place', stateSelector: '.js-add-comment-state', zipCodeSelector: '.js-add-comment-zipcode', countrySelector: '.js-add-comment-country', countryErrorSelector: '.comment-enter-country', }); }); }); $(document).ready(function () { var reportIds = { "451334": 0, "451335": 0, "451332": 0, "451333": 0, "451336": 0, "451337": 0, "400534": 0, "400535": 0, "400532": 0, "400533": 0, "460319": 0, "460318": 0, "400540": 0, "460317": 0, "400541": 0, "460316": 0, "400538": 0, "460315": 0, "400539": 0, "460314": 0, "400536": 0, "460313": 0, "400537": 0, "450599": 18, "460322": 4, "460321": 0, "460320": 0, "450600": 0, "403511": 0, "403518": 0, "403519": 0, "403516": 0, "403517": 0, "403514": 0, "403515": 11, "403512": 0, "403513": 0, "403520": 0, "409822": 11, "409823": 0, "409830": 0, "409831": 0, "409828": 0, "409829": 0, "409826": 0, "409827": 0, "409824": 0, "409825": 0, "406766": 0, "406767": 0, "406764": 0, "406765": 0, "406762": 0, "451306": 0, "406763": 0, "451318": 0, "451319": 0, "451317": 0, "406770": 0, "406771": 0, "406768": 0, "406769": 0, "451320": 0, }; $(".referee-response-container,.js-referee-report").each(function(index, el) { var reportId = $(el).attr("data-reportid"), reportCount = reportIds[reportId] || 0; $(el).find(".comments-count-container,.js-referee-report-views").html(reportCount); }); var uuidInput = $("#article_uuid"), oldUUId = uuidInput.val(), newUUId = "3474a9ca-21c1-400e-938e-6541270b8463"; uuidInput.val(newUUId); $("a[href*='article_uuid=']").each(function(index, el) { var newHref = $(el).attr("href").replace(oldUUId, newUUId); $(el).attr("href", newHref); }); }); An innovative open access publishing platform offering rapid publication and open peer review, whilst supporting data deposition and sharing. Browse Gateways Collections How it Works Contact For Developers Cookie Notice Privacy Notice RSS Submit Your Research Follow us © 2012-2026 F1000 Research Ltd. ISSN 2046-1402 | Legal | Partner of Research4Life • CrossRef • ORCID • FAIRSharing R.templateTests.simpleTemplate = R.template(' $text $text $text $text $text '); R.templateTests.runTests(); var F1000platform = new F1000.Platform({ name: "f1000research", displayName: "F1000Research", hostName: "f1000research.com", id: "1", editorialEmail: "
[email protected]", infoEmail: "
[email protected]", usePmcStats: true }); $(function(){R.ui.dropdowns('.dropdown-for-authors, .dropdown-for-about, .dropdown-for-myresearch');}); // $(function(){R.ui.dropdowns('.dropdown-for-referees');}); $(document).ready(function () { if ($(".cookie-warning").is(":visible")) { $(".sticky").css("margin-bottom", "35px"); $(".devices").addClass("devices-and-cookie-warning"); } $(".cookie-warning .close-button").click(function (e) { $(".devices").removeClass("devices-and-cookie-warning"); $(".sticky").css("margin-bottom", "0"); }); $("#tweeter-feed .tweet-message").each(function (i, message) { var self = $(message); self.html(linkify(self.html())); }); $(".partner").on("mouseenter mouseleave", function() { $(this).find(".gray-scale, .colour").toggleClass("is-hidden"); }); }); Sign In Remember me Forgotten your password? Sign In Cancel Email or password not correct. Please try again Please wait... $(function(){ // Note: All the setup needs to run against a name attribute and *not* the id due the clonish // nature of facebox... $("a[id=googleSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("GOOGLE"); $("form[id=oAuthForm]").submit(); }); $("a[id=facebookSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("FACEBOOK"); $("form[id=oAuthForm]").submit(); }); $("a[id=orcidSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("ORCID"); $("form[id=oAuthForm]").submit(); }); }); If you've forgotten your password, please enter your email address below and we'll send you instructions on how to reset your password. The email address should be the one you originally registered with F1000. Email address not valid, please try again You registered with F1000 via Google, so we cannot reset your password. To sign in, please click here . If you still need help with your Google account password, please click here . You registered with F1000 via Facebook, so we cannot reset your password. To sign in, please click here . If you still need help with your Facebook account password, please click here . Code not correct, please try again Reset password Cancel Email us for further assistance. Server error, please try again. If your email address is registered with us, we will email you instructions to reset your password. If you think you should have received this email but it has not arrived, please check your spam filters and/or contact for further assistance. Please wait... Register $(document).ready(function () { signIn.createSignInAsRow($("#sign-in-form-gfb-popup")); $(".target-field").each(function () { var uris = $(this).val().split("/"); if (uris.pop() === "login") { $(this).val(uris.toString().replace(",","/")); } }); });
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.