A Hybrid NLP Framework for Measuring Therapist Empathy in Online Counseling Responses | Research Square window.SnipcartSettings = { analytics: { enabled: false } }; (function() { var accessVector = localStorage.getItem('access_vector') || ''; window.dataLayer = window.dataLayer || []; if (accessVector) { window.dataLayer.push({ user: { profile: { profileInfo: { snid: accessVector } } } }); } })(); (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src='https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-K279D39R'); Browse Preprints In Review Journals COVID-19 Preprints AJE Video Bytes Research Tools Research Promotion AJE Professional Editing AJE Rubriq About Preprint Platform In Review Editorial Policies Our Team Advisory Board Help Center Sign In Submit a Preprint Cite Share Download PDF Research Article A Hybrid NLP Framework for Measuring Therapist Empathy in Online Counseling Responses Muhammad Hassam Aslam Khan This is a preprint; it has not been peer reviewed by a journal. https://doi.org/ 10.21203/rs.3.rs-9667616/v1 This work is licensed under a CC BY 4.0 License Status: Posted Version 1 posted You are reading this latest preprint version Abstract Empathy is central to effective therapeutic communication, but measuring empathy in large-scale online counseling interactions remains challenging. This paper presents a hybrid natural language processing framework for estimating therapist empathy in online counseling responses. We define empathy using linguistic and semantic markers such as emotional acknowledgment, validation, supportiveness, reflective listening, reassurance, and non-directive language. We then fine-tune a BERT-based classifier to identify empathetic expressions and combine its output with sentiment scores and normalized response length to construct an interpretable empathy score. The framework is applied to therapist responses from the CounselChat dataset, which contains real online counseling questions and therapist answers. We evaluate the model using classifier performance, compare the generated empathy scores with ChatGPT-based scores as a consistency check, and examine whether estimated empathy is associated with user engagement measures such as views and upvotes. The results suggest that transformer-based models can capture meaningful empathy-related language patterns in therapist responses, while the weak relationship between empathy scores and engagement indicates that platform metrics capture only part of perceived therapeutic quality. The study highlights the potential of NLP methods for large-scale analysis of therapeutic communication while emphasizing the need for expert clinical annotations and stronger validation in future work. Computer Architecture and Engineering Full Text Additional Declarations The authors declare no competing interests. Cite Share Download PDF Status: Posted Version 1 posted You are reading this latest preprint version Research Square lets you share your work early, gain feedback from the community, and start making changes to your manuscript prior to peer review in a journal. As a division of Research Square Company, we’re committed to making research communication faster, fairer, and more useful. We do this by developing innovative software and high quality services for the global research community. Our growing team is made up of researchers and industry professionals working together to solve the most critical problems facing scientific publishing. Also discoverable on Platform About Our Team In Review Editorial Policies Advisory Board Help Center Resources Author Services Accessibility API Access RSS feed Manage Cookie Preferences © Research Square 2026 | ISSN 2693-5015 (online) Privacy Policy Terms of Service Do Not Sell My Personal Information {"props":{"pageProps":{"initialData":{"identity":"rs-9667616","acceptedTermsAndConditions":true,"allowDirectSubmit":true,"archivedVersions":[],"articleType":"Research Article","associatedPublications":[],"authors":[{"id":637566999,"identity":"e0f0615e-f833-40f5-8f24-2d9524e43179","order_by":0,"name":"Muhammad Hassam Aslam Khan","email":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAZAAAAAyAQMAAABI0h/eAAAABlBMVEX///8AAABVwtN+AAAACXBIWXMAAA7EAAAOxAGVKw4bAAABB0lEQVRIiWNgGAWjYBACAxiDj4GBmRlIy4E4Bx4Qo4UNqsUYrCWBFC2JDSAePi3m7Mcff/iZY2fPxn462big5k76/LDDD4G22MnpNmDXYtmTYybZuy05sY0nd3PyjGPPcjfeTjMAakk2NjuAw2EHctgYeLcxJ7Ax5G4+zMN2OHfj7ASQlgOJ23BpOf/88ce/2+rt2fjfArX8O5xuODv9A34tNxIMpHm3HWZskwA6jLftcIK8dA4BW268MZOW3XY8sU3i7WZj3r7DhhukcwoOJBjg8cv59Mcf326rtufnz90szfPtsLz87PTNHz5U2Mnh0oItQMAkscpBQL6BFNWjYBSMglEwEgAAS2RjrFFDRToAAAAASUVORK5CYII=","orcid":"","institution":"Indaina University Indianapolis","correspondingAuthor":true,"prefix":"","firstName":"Muhammad","middleName":"Hassam Aslam","lastName":"Khan","suffix":""}],"badges":[],"createdAt":"2026-05-10 05:11:43","currentVersionCode":1,"declarations":{"humanSubjects":false,"vertebrateSubjects":false,"conflictsOfInterestStatement":false,"humanSubjectEthicalGuidelines":false,"humanSubjectConsent":false,"humanSubjectClinicalTrial":false,"humanSubjectCaseReport":false,"vertebrateSubjectEthicalGuidelines":false},"doi":"10.21203/rs.3.rs-9667616/v1","doiUrl":"https://doi.org/10.21203/rs.3.rs-9667616/v1","draftVersion":[],"editorialEvents":[],"editorialNote":"","failedWorkflow":false,"files":[{"id":109039653,"identity":"cc5e3359-8028-459d-9cbb-c3b105cf1ee9","added_by":"auto","created_at":"2026-05-12 03:32:25","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":368442,"visible":true,"origin":"","legend":"","description":"","filename":"THerapistempathy.pdf","url":"https://assets-eu.researchsquare.com/files/rs-9667616/v1_covered_40f0516c-fa58-4cf2-b3c1-bcbfb8088089.pdf"}],"financialInterests":"The authors declare no competing interests.","formattedTitle":"\u003cp\u003eA Hybrid NLP Framework for Measuring Therapist Empathy in Online Counseling Responses\u003c/p\u003e","fulltext":[],"fulltextSource":"","fullText":"","funders":[],"hasAdminPriorityOnWorkflow":false,"hasManuscriptDocX":false,"hasOptedInToPreprint":true,"hasPassedJournalQc":"","hasAnyPriority":true,"hideJournal":true,"highlight":"","institution":"","isAcceptedByJournal":false,"isAuthorSuppliedPdf":true,"isDeskRejected":"","isHiddenFromSearch":false,"isInQc":false,"isInWorkflow":false,"isPdf":true,"isPdfUpToDate":true,"isWithdrawnOrRetracted":false,"journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true},"keywords":"","lastPublishedDoi":"10.21203/rs.3.rs-9667616/v1","lastPublishedDoiUrl":"https://doi.org/10.21203/rs.3.rs-9667616/v1","license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"manuscriptAbstract":"\u003cp\u003eEmpathy is central to effective therapeutic communication, but measuring empathy in large-scale online counseling interactions remains challenging. This paper presents a hybrid natural language processing framework for estimating therapist empathy in online counseling responses. We define empathy using linguistic and semantic markers such as emotional acknowledgment, validation, supportiveness, reflective listening, reassurance, and non-directive language. We then fine-tune a BERT-based classifier to identify empathetic expressions and combine its output with sentiment scores and normalized response length to construct an interpretable empathy score. The framework is applied to therapist responses from the CounselChat dataset, which contains real online counseling questions and therapist answers. We evaluate the model using classifier performance, compare the generated empathy scores with ChatGPT-based scores as a consistency check, and examine whether estimated empathy is associated with user engagement measures such as views and upvotes. The results suggest that transformer-based models can capture meaningful empathy-related language patterns in therapist responses, while the weak relationship between empathy scores and engagement indicates that platform metrics capture only part of perceived therapeutic quality. The study highlights the potential of NLP methods for large-scale analysis of therapeutic communication while emphasizing the need for expert clinical annotations and stronger validation in future work.\u003c/p\u003e","manuscriptTitle":"A Hybrid NLP Framework for Measuring Therapist Empathy in Online Counseling Responses","msid":"","msnumber":"","nonDraftVersions":[{"code":1,"date":"2026-05-12 03:32:03","doi":"10.21203/rs.3.rs-9667616/v1","editorialEvents":[{"type":"communityComments","content":0}],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true}}],"origin":"","ownerIdentity":"36384000-91f7-44d7-ae5d-68a0a0f7ce39","owner":[],"postedDate":"May 12th, 2026","published":true,"recentEditorialEvents":[],"rejectedJournal":[],"revision":"","amendment":"","status":"posted","subjectAreas":[{"id":67861164,"name":"Computer Architecture and Engineering"}],"tags":[],"updatedAt":"2026-05-12T03:32:03+00:00","versionOfRecord":[],"versionCreatedAt":"2026-05-12 03:32:03","video":"","vorDoi":"","vorDoiUrl":"","workflowStages":[]},"version":"v1","identity":"rs-9667616","journalConfig":"researchsquare"},"__N_SSP":true},"page":"/article/[identity]/[[...version]]","query":{"redirect":"/article/rs-9667616","identity":"rs-9667616","version":["v1"]},"buildId":"XKTyCvWXoU3ODBz1xrDgd","isFallback":false,"isExperimentalCompile":false,"dynamicIds":[84888],"gssp":true,"scriptLoader":[]}
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.