Integrating Group and Individual Fairness in Clinical AI: A Post-Hoc, Model-Agnostic Framework for Fairness Auditing

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Ensuring fairness across diverse patient populations is a fundamental challenge for clinical AI systems, yet current fairness evaluation approaches create critical blind spots. Group-level metrics capture systemic disparities but miss patient-level variations, while individual fairness frameworks ensure consistency but potentially obscure structural biases. In this paper, we propose EquiLense, a post-hoc, model-agnostic framework that bridges these perspectives through clinical similarity matching and comprehensive fairness auditing. Our method introduces the Mean Predicted Probability Difference (MPPD), which quantifies prediction inconsistencies between clinically similar patients across demographic groups, integrating both individual-level consistency and group-level equity assessment. Moreover, we provide flexible similarity matching using clinical features and comprehensive visualization tools that support practical deployment in healthcare settings. Applied to electronic health record data from over 59,000 surgical patients, our framework revealed disparities in prediction consistency even when overall model performance appeared strong. EquiLense identified differences in predicted probabilities between clinically similar patients from different racial groups, disparities that were substantially reduced when sensitive attributes were excluded from model training. Our method provides a clinically relevant and interpretable approach to fairness auditing that enables healthcare practitioners to identify, understand, and address algorithmic disparities in real-world deployment settings.
Full text 53,992 characters · extracted from preprint-html · click to expand
Integrating Group and Individual Fairness Auditing in Clinical AI: A Post-Hoc, Model-Agnostic Approach | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Integrating Group and Individual Fairness Auditing in Clinical AI: A Post-Hoc, Model-Agnostic Approach View ORCID Profile Javen Xu , View ORCID Profile Yeon-Mi Hwang , Samhita Kondareddy , Inés Dormoy , Serena Liang Jing , View ORCID Profile Malvika Pillai , View ORCID Profile Catherine Curtin , View ORCID Profile Tina Hernandez-Boussard doi: https://doi.org/10.1101/2025.09.03.25334999 Javen Xu 1 Department of Medicine (Computational Medicine), Stanford University , Stanford, CA, USA 2 Department of Biomedical Data Science, Stanford University , Stanford, CA, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Javen Xu Yeon-Mi Hwang 1 Department of Medicine (Computational Medicine), Stanford University , Stanford, CA, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Yeon-Mi Hwang For correspondence: ymh{at}stanford.edu Samhita Kondareddy 1 Department of Medicine (Computational Medicine), Stanford University , Stanford, CA, USA 3 Department of Computer Science, School of Engineering, Stanford University , Stanford, CA, USA BS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Inés Dormoy 1 Department of Medicine (Computational Medicine), Stanford University , Stanford, CA, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Serena Liang Jing 4 School of Medicine, Stanford University , Stanford, CA, USA BS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Malvika Pillai 1 Department of Medicine (Computational Medicine), Stanford University , Stanford, CA, USA 5 Veterans Affairs Palo Alto Health Care System , Palo Alto, CA, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Malvika Pillai Catherine Curtin 6 Department of Surgery, Veterans Affairs Palo Alto Health Care System , Palo Alto, CA, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Catherine Curtin Tina Hernandez-Boussard 1 Department of Medicine (Computational Medicine), Stanford University , Stanford, CA, USA 2 Department of Biomedical Data Science, Stanford University , Stanford, CA, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tina Hernandez-Boussard Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Objectives To develop and demonstrate a practical post-hoc, model-agnostic fairness auditing tool that integrates group-level and individual-level fairness assessment for clinical prediction models. Materials and Methods We developed EquiLense, a fairness auditing tool that operationalizes three components: group fairness evaluation using established and novel group-level disparity metrics, individual fairness assessment, and the Mean Predicted Probability Difference (MPPD), a novel metric quantifying prediction inconsistencies between clinically similar patients across demographic groups. We applied EquiLense to EHR data from 59,047 surgical patients across post-surgical delirium and 30-day readmission prediction models, and demonstrated MPPD on two external benchmarks with documented bias: the COMPAS recidivism dataset and the UCI Adult Income dataset. Results Applied to delirium prediction (AUROC 0.78), MPPD identified prediction inconsistencies between clinically similar patients that varied systematically across demographic groups. Among patients clinically similar to White patients, Asian patients exhibited the largest disparity (MPPD difference = 0.045). In external benchmarks, MPPD assigned the highest disparity scores to African American defendants in COMPAS and to female workers in UCI Adult Income, consistent with documented bias in both datasets. Discussion Current fairness evaluation approaches address group-level disparities or individual-level consistency in isolation, creating blind spots in algorithmic fairness assessment. EquiLense bridges this gap by integrating established fairness metrics with MPPD in a single post-hoc tool, enabling comprehensive and interpretable fairness auditing without model retraining. Conclusion EquiLense provides a practical approach to fairness auditing in clinical AI by combining group and individual fairness perspective through MPPD, a novel metric demonstrated on both clinical and benchmark datasets. Lay Summary Clinical prediction models are increasingly used across clinical care, research, and decision-making, but they can produce unfair outcomes for patients. Fairness can be evaluated at the group level, by comparing outcomes across demographic categories, or at the individual level, by assessing whether clinically similar patients receive consistent predictions. Most existing tools address one perspective but not both together. We developed EquiLense, a post-hoc fairness auditing tool that bridges these two perspectives. EquiLense operationalizes three components: (1) group fairness evaluation using using multiple group-level disparity metrics; (2) individual fairness assessment measuring prediction consistency among clinically similar patients; and (3) the Mean Predicted Probability Difference (MPPD), a novel metric that quantifies how differently a model treats two patients who are clinically similar but belong to different demographic groups. We applied EquiLense to surgical outcome prediction models using data from over 59,000 patients, and validated MPPD on two external datasets with well-documented bias (COMPAS recidivism and UCI Adult Income). MPPD detected known disparities in both benchmarks. In our clinical application, we found that models can achieve strong overall performance while still producing systematically different predictions for clinically similar patients from different sensitive attributes. EquiLense requires no model retraining, applies to any clinical prediction model, and produces interpretable outputs grounded in clinically familiar features, making it directly applicable in real-world healthcare settings. 1. BACKGROUND AND SIGNIFICANCE As artificial intelligence (AI) becomes increasingly integrated in clinical decision-making, ensuring algorithmic fairness is both an ethical priority and a practical challenge.[ 1 – 5 ] Current fairness evaluations are fragmented, focusing either on group-level disparities (e.g. across demographic categories) or individual level inconsistencies.[ 2 , 6 ] This creates blind spots. Group metrics can obscure important patient-level variations, while individual metrics may fail to reveal systemic biases affecting entire demographic groups.[ 7 ] Group-level metrics are valuable for identifying systemic disparities that disadvantage marginalized populations. However, they cannot reveal whether clinically similar individuals across demographic groups receive consistent predictions.[ 2 , 8 , 9 ] Individual fairness frameworks address this gap by ensuring prediction consistency among comparable patients. Dwork et al.[ 10 ] define individual fairness as a constraint that any two similar individuals, as defined by a task-specific metric over observable features, should receive similar predictions. This is distinct from group fairness. A model can show no group-level disparity while still treating clinically similar individuals inconsistently across demographic groups. Although it is intuitively appealing, individual fairness has seen limited adoption in healthcare, where research remains in its early stages.[ 11 ] A key challenge is that defining a clinically meaningful similarity metric is inherently subjective, as different clinical contexts and tasks may require different definitions of similarity. Additionally, data-driven similarity metrics risk encoding existing biases, which may perpetuate rather than reduce health inequities.[ 11 , 12 ] Despite growing recognition of its importance, practical tools for systematically quantifying individual-level fairness in healthcare remain limited.[ 5 ] Counterfactual fairness (CF) operationalizes a related but distinct principle through a causal lens, asking whether a prediction would change if the sensitive attribute were different, but requires causal graph specification or synthetic counterpart generation that is difficult to justify in EHR settings.[ 11 , 13 – 15 ] Recognizing that group and individual fairness each capture only partial perspectives, several approaches have sought to integrate both. Speicher et al.[ 16 ] introduced a generalized entropy-based metric that decomposes total unfairness into between-group and within-group components, offering a unified measure of disparity. Xu and Strohmer [ 6 ] presented a theoretical framework that characterizes the conditions under which group and individual fairness are compatible, but without operational implementation. FairGI jointly optimizes both using adversarial learning for graph neural networks.[ 17 ] In our prior work, FairEHR-CLP[ 18 ] combines CF-inspired individual fairness through synthetic counterpart generation with group fairness evaluation, jointly optimized during model training using contrastive learning. However, some of these approaches lack operational implementation, while others require model retraining or are tied to specific architectures, limiting their applicability for auditing already-deployed clinical models, a gap that EquiLense is designed to fill. We introduce EquiLense, a post-hoc, model-agnostic fairness auditing tool that integrates group fairness evaluation, individual fairness assessment, and MPPD, a novel metric designed to bridge group-level disparity detection and individual-level prediction consistency, into a single accessible tool. Existing metrics each capture important but partial perspectives on fairness. EquiLense brings them together, with MPPD as its methodological contribution. MPPD quantifies prediction inconsistencies between real, observed patients who are clinically similar but belong to different demographic groups. It operationalizes consistency-based individual fairness by grounding similarity in observable clinical features using cosine similarity, making its assumptions explicit and directly adjustable by practitioners.[ 10 , 11 ] We demonstrate EquiLense using real-world surgical outcome prediction models and evaluate MPPD on two external benchmarks with known bias. 2. MATERIALS AND METHODS 2.1 Study design and population This study was approved by the Institutional Review Board at Stanford University (Stanford, CA, USA; Protocol IRB-34551) with a waiver of informed consent. We applied Equilense to EHR data from an integrated healthcare system operating on a unified Epic platform. The cohort included 59,047 surgical patients treated between January 1, 2012 to December 1, 2022. The cohort included patients aged 50 years or older who had inpatient stays of 90 days or less and survived at least 30 days post-surgery. Our analysis incorporated all post-surgical clinical documentation, including progress and nursing notes. For patients with multiple surgeries, only the initial procedure was included. Cohort details are available in Table 1 , Figure S1 and described in detail elsewhere.[ 19 ] View this table: View inline View popup Table 1. Baseline characteristics of Surgical Patients 2.2 Model development 2.2.1 Outcome The primary outcome was post-surgical delirium, identified using ICD diagnostic codes (Table S1), documented positive Confusion Assessment Method (CAM) scores, and natural language processing of clinical notes (Method S1). We used NLP to supplement structured data because delirium is frequently underreported in coded fields. The secondary outcome was 30-day hospital readmission, defined as any inpatient admission within 30 days of discharge. 2.2.2 Features We categorized features into sensitive and clinical attributes. Sensitive attributes included sex, race/ethnicity, and insurance type (as a proxy for socioeconomic status). While sex can serve as either a sensitive or clinical attribute depending on context, we classified it as sensitive for this study. Clinical attributes included age, body mass index (BMI), pre-surgical pain score, Charlson Comorbidity Index (CCI), opioid-naïve status, mental health diagnoses, anesthesia type, and surgery type. We used the most recent BMI recorded prior to surgery. Pre-surgical pain scores were collected from a 180-day window before surgery through the day of surgery. The CCI was calculated using diagnoses from the two years preceding surgery. For descriptive analyses, we categorized CCI scores as low (0–2), moderate (3– 4), or high (≥5), while continuous scores were used in machine learning models. We defined opioid-naïve status as having no opioid prescription in the 6 months before surgery. Mental health diagnoses were identified using the five most frequent ICD-10 F-code categories in our cohort: mental organic disorders, substance use disorders, mood disorders, anxiety disorders, and behavioral symptoms (Table S1). Anesthesia type was determined through rules-based natural language processing of operative and anesthesia procedure notes, with cases involving both general and regional anesthesia categorized as hybrid. Surgery type was extracted from surgical notes (Method S2). All features were preprocessed using standard normalization and one-hot encoding. Missing values were imputed using median values. 2.2.3 Models We used an 80/20 train-test split. To address class imbalance, we applied the Synthetic Minority Oversampling Technique (SMOTE) to the training set. Models were developed using logistic regression and extreme gradient boosting (XGB). Model performance was assessed using accuracy, precision, recall, and area under the receiver operating characteristic curve (AUROC). All analyses were conducted using scikit-learn (v1.3.0) and XGBoost (v2.1.1) in Python. We also evaluated the contribution of each sensitive attribute to model performance and fairness by training models with and without individual sensitive attributes and examining the resulting disparities. We note that the models were not optimized for predictive performance or clinical deployment but serve as demonstration cases to illustrate the application of the EquiLense fairness auditing framework. 2.3 Fairness evaluation 2.3.1 Fairness evaluation visualization framework Group fairness was evaluated using Demographic Parity Difference (DPD), Equal Opportunity Difference (EOD), Equalized Odds (EO),[ 20 ] as well as the Error Distribution Disparity Index (EDDI), a novel group fairness metric designed for clinical settings with diverse group sizes and class imbalance.[ 18 ] Individual fairness was assessed by measuring prediction consistency among clinically similar patients, with similarity defined using distance-based metrics. We calculated individual fairness metrics with both a narrow set of clinical variables (age, CCI, BMI) and the full feature set to examine how the definition of clinical similarity influences fairness estimates. For the full feature set, comorbidities included in the CCI were one-hot encoded. All metrics were computed on the test set. Currently, there is currently no consensus on how to define clinical similarity in individual fairness assessments. Similarity may vary depending on the context, population, and prediction task. To support flexible exploration, our codebase allows users to specify their own similarity criteria, including variables, distance metrics, and thresholds. This enables customized fairness analyses across diverse clinical use cases. 2.3.2 Mean Predicted Probability Difference (MPPD) To evaluate fairness at both the group and individual levels, we developed a novel integrated metric, the Mean Predicted Probability Difference (MPPD). This metric quantifies inconsistencies in predicted probabilities between clinically similar patients who belong to different sensitive groups ( Figure 1 ). Download figure Open in new tab Figure 1. Conceptual overview of Mean Predicted Probability (MPPD) metric Illustration of how MPPD is calculated. For each patient, predicted probabilities are compared with those of clinically similar patients from a different demographic group. In this example, the sensitive attribute is race. These differences are then averaged across group pairs to quantify disparities in prediction consistency. We had two approaches: 1) reference-based and 2) pairwise. The reference-based approach begins by identifying the largest group within each sensitive attribute (e.g., Non-Hispanic White patients for race/ethnicity) as the reference group. For each patient in the reference group, we identify clinically similar individuals purely based on cosine similarity computed on clinical attributes. The similarity threshold, set to 0.01 by default but user-selectable, defines matched pairs. The set of clinical and social attributes used to define similarity can also be tailored based on user judgment. For example, for a Non-Hispanic White patient, they could have several matched pairs as long as the other patient’s clinical similarity distance is within 0.01 (the similar patients could be Non-Hispanic White patients, Non-Hispanic Asian patients, Hispanic patients, or from any of the race/ethnicity groups). They might also have no matched pair if there is no other patient within 0.01 of clinical similarity distance. For each matched pair, we calculate the absolute difference in model predicted probabilities, and we aggregate these differences by group (for all of reference and non-reference groups) to obtain the average MPPD value for each group. This process captures whether patients from non-reference groups receive systematically different predictions despite having similar clinical profiles, thereby integrating elements of group fairness (via inter-group comparison) and individual fairness (via intra-clinical similarity). To complement the reference-based approach and avoid assumptions about any one group being “fair” by default, we extended MPPD to a pairwise version. In this version, we performed bi-directional matching between all combinations of sensitive attribute groups, allowing identification of clinically similar patients across group-pairs (e.g., Black-White, Asian-Hispanic). For each group-pair, we computed the average absolute difference in predicted probabilities between matched patients, producing a matrix of pairwise MPPD scores. This symmetric comparison allows for a more comprehensive understanding of prediction consistency across all groups, including disparities between minority populations that would be overlooked in a reference-only framework. Together, these two implementations of MPPD provide a flexible and interpretable method for uncovering fairness concerns in clinical prediction models. The interpretability of the framework comes from its use of clinically familiar features already involved in model development, while its flexibility lies in allowing users to define clinical similarity according to their own context (e.g., age, comorbidity burden, surgical type). This alignment with routine clinical reasoning makes the outputs easier for practitioners to understand and apply in practice. 2.4 External Benchmark Validation To demonstrate MPPD against known ground truth, we applied it to two external datasets selected for their well-documented algorithmic bias: the COMPAS recidivism dataset (n=6172) and the UCI Adult Income dataset (n=32,561).[ 21 , 22 ] For each dataset, an XGBoost classifier was trained on an 80/20 train-test split as a demosntration case. For COMPAS, the reference group was White defendants and similarity features included number of prior offenses, age bracket, and misdemeanor status. For UCI Adult Income, reference groups were White and Male. Similarity features included age, education level, capital gain, capital loss, hours worked per week, occupation, and work class. 3. RESULTS 3.1 Descriptive statistics The cohort comprised 59,047 surgical patients who underwent 75,179 surgeries at a large academic medical center ( Table 1 and Figure S1). Post-surgical delirium prevalence was 14.9% and 30-day readmission was 8.8%. The median age at surgery was 66 years, with 52% of patients being female. Most patients identified as Non-Hispanic White (63%), followed by Non-Hispanic Asian (13%) and Hispanic (12%). At surgery, 59% had private insurance ( Table 1 ). 3.2 Model performance For delirium prediction, the XGBoost model achieved an accuracy of 0.84, AUROC of 0.78, precision of 0.45, and recall of 0.26 (Table S2). Removing individual sensitive attributes (race/ethnicity, sex, or insurance) had minimal impact on model performance, with accuracy and AUROC remaining stable at 0.83 and 0.77, respectively. For 30-day readmission, XGBoost achieved an accuracy of 0.91, AUROC of 0.61, precision of 0.19, and recall of 0.02; excluding sensitive attributes again had negligible effects. Logistic regression performed worse across both outcomes. In both models, sensitive attributes had low-to-moderate feature importance and minimal effect on predictive performance when removed (Figure S2-S3). 3.3 Group and Indiviudal Fairness Figure 2a presents a radar plot summarizing group fairness metrics for the delirium prediction model, using race as the sensitive attribute. Both maximum and average values are presented for DPD, EOD, EO, and EDDI, with disparity thresholds set at 0.01 (low), 0.05 (moderate), and 0.1 (high). EDDI shows the largest disparity, reflecting differences in prediction error distributions across groups. Figure 2b presents EO difference values by race and ethnicity. Asian and non-Hispaic other patients showed the highest EO differences (0.0389), in the moderate disparity range, while White patients had the lowest (0.0096). Download figure Open in new tab Figure 2. Visual summary of group fairness metrics in the delirium prediction model Abbreviations: DPD, Demographic Parity Difference; EOD, Equal Opportunity Difference; EDDI, Error Distribution Disparity Index. Panel A displays a radar chart summarizing group fairness metrics across demographic groups, including DPD, EOD, Equalized Odds, and EDDI. Dashed lines indicate thresholds for low (0.01), medium (0.05), and high (0.1) disparity. Panel B presents Equalized Odds differences by race and ethnicity, highlighting subgroup disparities. Asian patients showed the highest disparity, and all groups fell within the low or moderate range. Test sample sizes and disparity values are shown for each group. Figure 3 shows individual fairness gaps across feature sets and distance thresholds. Models using the full set of clinical features consistently achieved lower individual fairness gaps than narrower feature sets. These results illustrate how similarity definition influences fairness estimates, and presented as components of EquiLense’s integrated auditing toolkit. Download figure Open in new tab Figure 3. Individual fairness Scores by feature set and distance threshold This displays individual fairness scores across varying clinical feature sets and distance thresholds. Individual fairness was assessed by calculating the average difference in predicted probabilities between clinically similar patients, with lower values indicating greater consistency (i.e., better fairness). Models using a comprehensive set of clinical features, including surgery and anesthesia type, showed lower fairness scores compared to those using narrower feature sets (such as age, BMI, Charlson Comorbidity Index, or comorbidity features alone). 3.4 Mean Predicted Probability Difference To demonstrate how EquiLense can inform practical modeling decisions, we examined whether excluding sensitive attributes from model training affects MPPD scores. Figure 4a presents MPPD values for the delirium model across sensitive attributes group. Among patients clinically similar to a White patient, Asian patients exhibited the largest prediction disparity (MPPD White-Asian = 0.119, MPPD White-White = 0.074, □ = 0.045), indicating lower individual fairness for that group. Excluding race and ethnicity from model training reduced these disparities across most group pairs (Figure S4), suggesting that feature inclusion directly influences prediction consistency between demographically different but clinically similar patients. Download figure Open in new tab Figure 4. MPPD across sensitive attribute groups Panel A shows MPPD values comparing predicted probabilities among clinically similar patients across sensitive attribute groups, including race/ethnicity, sex, and insurance type. The largest disparity was observed between Asian and White patients (White-Asian: 0.119 vs. White-White: 0.074; Δ = 0.045), highlighting prediction inconsistencies. Panel B displays the full pairwise MPPD matrix across racial and ethnic groups, with each cell reflecting the MPPD between matched patients from the corresponding group pair. Within-group MPPD values are non-zero because each patient is compared to another clinically similar patient from the same group. Figure 4b presents the full pairwise MPPD matrix. Within-group MPPD values (e.g. White-White) are non-zero because each patient is compared to another clinically similar patient from the same group. When race and ethnicity were excluded from model features (Figure S5), MPPD scores decreased substantially across most group pairs. The exception was the Black-Native American pair; this estimate should be interpreted with caution given the small Native American test sample size (n=32), a recognized limitation in health disparities research. Similar trends were observed in the 30-day readmission model. 3.5 External Benchmark Validation MPPD detected known bias in both external datasets. In COMPAS, African American defendants received the highest MPPD score (0.1122), consistent with the well-documented pattern of disproportionately elevated recidivism risk scores assigned to Black defendants relative to White defendants with similar criminal histories.[ 23 , 24 ] In the UCI Adult Income dataset, female workers received the highest MPPD score (0.2363), consistent with documented gender bias in income prediction.[ 25 ] Across both datasets, MPPD assigned the highest disparity scores to groups with known algorithmic disadvantage, supporting its validity as a fairness screening metric. 4. DISCUSSION We present EquiLense, a model-agnostic, post-hoc auditing tool that addresses a gap in current fairness evaluation: the lack of a practical, accessible tool that integrates both group-level and individual-level fairness assessment for already-deployed clinical prediction models. While existing tools address one perspective or require model retraining, EquiLense integrates group fairness evaluation, individual fairness assessment, and MPPD, its novel methodological contribution, into a single framework that requires no model retraining and applies to any clinical prediction model. Applied to surgical outcome models, we demonstrate that strong overall predictive performance does not preclude systematic prediction inconsistencies across demographic groups. External benchmark demonstration on COMPAS and UCI Adult Income confirms that MPPD detects meaningful disparities when bias is known to exist. Current fairness evaluation approaches remain fragmented. Although fairness and bias mitigation strategies in clinical AI are increasingly studied, consensus on what constitutes fairness remains elusive, with definitions that often conflict and require trade-offs. [ 5 ] Different fairness metrics serve different purposes, and no single metric is sufficient on its own. Group fairness metrics identify systemic disparities but are insensitive to individual-level variation within groups, while individual fairness metrics have seen limited adoption in healthcare because similarity definitions are inherently context-dependent, varying across clinical tasks, populations, and outcomes.[ 11 ] Evaluating fairness comprehensively requires applying multiple metrics that reflect different fairness perspectives, as different metrics can lead to conflicting conclusions and the choice of metric should align with the clinical context and ethical goals of the specific use case.[ 5 ] EquiLense addresses a key dimension of this fragmentation by integrating group fairness evaluation, individual fairness assessment, and MPPD into a single tool, enabling fairness assessment that captures both population-level patterns and patient-level inconsistencies. The tool allows users to specify clinically grounded similarity criteria, aligning the evaluation with clinical reasoning relevant to the prediction task at hand. MPPD values represent prediction differences between clinically similar patients, allowing users to assess both whether disparities exist and their magnitude, supporting practical decision-making about model deployment and feature selection. MPPD is designed to complement, not replace, existing fairness approaches. It compares real, observed patients who are already clinically similar, grounding the fairness assessment in observed data. Compared to CF-based approaches, MPPD is computationally straightforward and does not require causal group specification, synthetic data generation, or model training, lowering the barrier for practical fairness auditing in clinical settings. Its assumptions concern which clinical variables define meaningful similarity. These are explicit, clinically interpretable, and directly adjustable by practitioners. Studies suggest that fairness should be addressed throughout all phases of model development and deployment, not only at a single stage.[ 5 ] MPPD is particularly well-suited for post-hoc auditing of already-deployed models, a stage that training-time approaches cannot address. In this way, EquiLense and FairEHR-CLP,[ 18 ] our CF-inspired in-processing method, address complementary stages of the clinical AI lifecycle. FairEHR-CLP evaluates and addresses fairness during model development, while EquiLense detects and surfaces disparities after deployment through MPPD, its individual-to-group fairness bridging metric. MPPD is intended as a screening and auditing metric. A high MPPD between two demographic groups indicates that clinically similar patients from those groups receive systematically different predictions. Whether this reflects algorithmic bias, an imperfect similarity specification, or clinically appropriate distinctions requires further investigation and clinical judgment. This interpretive challenge is shared with group fairness metrics, which similarly flag disparities without resolving their cause. EquiLense supports this investigative process by enabling sensitivity analyses across similarity thresholds, feature sets, and modeling choices. To illustrate how EquiLense can inform feature selection decisions, we examined the effect of excluding sensitive attributes from model training on MPPD scores. We observed that excluding sensitive attributes, such as race/ethnicity and insurance type, from model training resulted in lower MPPD scores, indicating improved fairness in prediction consistency across demographic groups. In our demonstration, removing these attributes reduced disparities between clinically similar patients without affecting overall model performance, though MPPD estimates for the Black-Native American pair should be interpreted with caution given the small Native American test sample size (n=32), a recognized limitation in health disparities research. These sensitive attributes also had low feature importance, suggesting minimal contribution to predictive accuracy. However, algorithmic bias can arise from multiple sources, including data imbalance, labeling variation, and broader structural inequities.[ 4 , 11 ] There is rarely a single solution. Removing a sensitive attribute is one possible approach but does not guarantee improved fairness in all cases.[ 5 ] EquiLense enables users to evaluate these trade-offs by testing how different modeling choices affect both fairness and performance. A key strength of EquiLense is its adaptability to diverse clinical contexts. Fairness goals and clinical priorities vary across healthcare settings, so the framework allows users to tailor evaluations accordingly. For example, age may be treated as a clinical predictor in surgical risk stratification but as a sensitive attribute in fairness assessments, while sex may serve as a biological covariate or as a sensitive attribute depending on the use case. Users can specify which variables define clinical similarity, which are treated as sensitive attributes, and adjust similarity thresholds and metric selection. This flexibility ensures fairness evaluation aligns with the specific priorities of each clinical application. Our approach is conceptually related to counterfactual fairness methods but differs in how it operationalizes the fairness principle. CF evaluates whether a model’s prediction would change for a hypothetical version of the same individual with a different demographic identity. This requires specification of a causal graph that captures how changes in sensitive attributes propagate through other features.[ 13 , 26 , 27 ] CF-inspired approaches, including our prior work FairEHR-CLP,[ 18 ] may approximate this through synthetic counterpart generation without full causal graph specification, but still require assumptions about which features would change under a different demographic identity.[ 14 ] Such assumptions are rarely verifiable in complex clinical settings due to unmeasured confounders, treatment heterogeneity, and documentation biases in EHR data. [ 14 , 15 ] While MPPD also relies on assumptions about which clinical variables appropriately define similarity, these are grounded in observable clinical features and directly adjustable by practitioners based on domain knowledge. We acknowledge that CF-based approaches may account for certain confounding pathways that MPPD does not by modeling causal structure. However, such causal assumptions are rarely verifiable in EHR settings, and concerns about structural bias in EHR-derived features apply equally to all EHR-based research. In contrast to CF, EquiLense matches real, observed patients from different demographic groups who are clinically similar, rather than hypothetical or synthetic counterparts. This grounds the fairness assessment in observed data, making its assumptions explicit and inspectable. A further strength of the pairwise comparison approach is that it enables symmetric evaluation across all group combinations without relying on a fixed reference group, surfacing disparities between minority populations that reference-based approaches would overlook. Despite its advantages, our approach has several limitations. First, MPPD is not a causal measure and cannot determine whether observed differences in predicted probabilities reflect unfair bias or clinically appropriate distinctions. It provides a systematic way to surface inconsistencies that warrant further investigation but requires clinical judgment to interpret. Second, fairness evaluation accuracy depends on how well clinical similarity is defined. Incomplete or noisy feature data may lead to mismatches that distort estimates. Defining similarity inherently involves subjective decisions about which features to include and how to weigh them, and fairness estimates can be sensitive to similarity thresholds. Additionally, chosen similarity variables such as CCI and BMI may themselves reflect structural inequities in healthcare access and documentation, a limitation shared by all EHR-based research. While EquiLense allows flexible specification of these parameters, users must carefully consider what constitutes meaningful similarity for their specific task. Future work should investigate how threshold selection affects fairness estimates and whether standard guidelines can be developed. Third, we did not explore interactions between demographic and clinical predictors, which may reveal additional sources of bias. Fourth, while EquiLense identifies disparities in model outputs, it does not assess downstream clinical consequences such as differences in care decisions or health outcomes. Finally, as a diagnostic tool, EquiLense does not include mitigation strategies. Future work should focus on validation of EquiLense on across a broader range of clinical prediction tasks and patient populations, development of integrated approaches combining fairness auditing with bias mitigation, and establishment of clinical thresholds for meaningful MPPD values across different prediction tasks. Further investigation into how similarity threshold selection affects fairness estimates, and whether standardized guidelines can be developed across clinical contexts, represents an important direction. Formal usability testing with clinical stakeholders would also be a valuable next step to evaluate whether EquiLense’s design choices effectively support practical fairness auditing in real-world healthcare settings. 5. CONCLUSION We introduced EquiLense, a flexible and model-agnostic tool for auditing fairness in clinical prediction models. By integrating group fairness evaluation, individual fairness assessment, and MPPD into a single post-hoc framework, EquiLense addresses a key gap in current fairness evaluation: the lack of a practical tool that captures both population-level disparities and patient-level prediction inconsistencies without requiring model retraining. Using post-surgical delirium and readmission models, we demonstrated how EquiLense surfaces fairness concerns, supports sensitivity analyses, and guides decisions about model design and feature inclusion. External validation on COMPAS and UCI Adult Income confirms that MPPD detects meaningful disparities when bias is known to exist. As clinical AI systems become more deeply integrated into care delivery, comprehensive and interpretable fairness auditing frameworks will be essential for ensuring that these systems advance rather than undermine health equity. Funding The project was supported by grant number R01HS024096 from the Agency for Healthcare Research and Quality. The content is solely the responsibility of the authors and do not necessarily represent the official views of the Agency for Healthcare Research and Quality. This study was conducted independently of the funding sources. Competing interests The authors declare no competing interests. Author contributions JX: Conceptualization, Resources, Data Curation, Software, Formal Analysis, Investigation, Visualization, Data Interpretation, Methodology, Writing - original draft, Writing - review and editing; Y-MH: Conceptualization, Resources, Data Curation, Software, Formal Analysis, Investigation, Visualization, Data Interpretation, Methodology, Writing - original draft, Writing - review and editing; SK: Formal Analysis, Investigation, Writing – original draft; ID: Methodology, Data Curation, Software, Writing - review and editing; SLJ: Investigation, Writing - review and editing; MP: Methodology, Data Interpretation, Writing - review and editing; CC: Data Curation, Data Interpretation, Methodology, Writing - review and editing; TH-B: Conceptualization, Supervision, Writing - review and editing, Administrative, technical, or material support; JX, Y-MH, ID, MP and TH-B had full access to the primary cohort data in the study and verified the data. All authors read and approved the final manuscript. Data availability The data underlying this article cannot be shared publicly due to patient privacy protections and institutional data use agreements. Deidentified data may be available upon reasonable request to the corresponding author and with appropriate data use agreements and institutional review board approvals. Code implementing the EquiLense framework is publicly available at https://github.com/su-boussard-lab/equilense_fairness Alt text Figure 1 :Alt text: Diagram illustrating how MPPD is calculated by comparing predicted probabilities between clinically similar patients from different racial groups, with differences averaged across group pairs to quantify prediction disparities. Figure 2 :Alt text: Panel A shows a radar chart summarizing group fairness metrics including DPD, EOD, EO, and EDDI for the delirium prediction model with disparity thresholds marked. Panel B shows a bar chart of Equalized Odds differences by race and ethnicity with Asian patients showing the highest disparity. Figure 3 :Alt text: Line graph showing individual fairness scores across different clinical feature sets and distance thresholds, with models using comprehensive feature sets showing lower fairness gaps than narrower feature sets. Figure 4 :Alt text: Panel A shows bar charts of MPPD values across race/ethnicity, sex, and insurance type sensitive attribute groups. Panel B shows a heatmap of pairwise MPPD scores across all racial and ethnic group combinations. Acknowledgments None Footnotes We have revised the entire manuscript. References [1]. ↵ Chen IY , Joshi S , Ghassemi M. Treating health disparities with artificial intelligence . Nat Med 2020 ; 26 : 16 – 17 . OpenUrl CrossRef PubMed [2]. ↵ Feng Q , Du M , Zou N , et al. Fair Machine Learning in Healthcare: A Review . arXiv [cs.LG] , http://arxiv.org/abs/2206.14397 ( 2022 , accessed 24 July 2025 ). [3]. Rajkomar A , Hardt M , Howell MD , et al. Ensuring fairness in machine learning to advance health equity . Ann Intern Med 2018 ; 169 : 866 – 872 . OpenUrl CrossRef PubMed [4]. ↵ Mehrabi N , Morstatter F , Saxena N , et al. A survey on bias and fairness in machine learning . arXiv [cs.LG] , http://arxiv.org/abs/1908.09635 ( 2019 , accessed 4 September 2025 ). [5]. ↵ van der Meijden SL , Wang Y , Arbous MS , et al. Navigating fairness in AI-based prediction models: Theoretical constructs and practical applications . medRxiv . Epub ahead of print 24 March 2025. DOI: 10.1101/2025.03.24.25324500 . OpenUrl Abstract / FREE Full Text [6]. ↵ Xu S , Strohmer T. On the (in)compatibility between group fairness and individual fairness . arXiv [math.ST] , http://arxiv.org/abs/2401.07174 ( 2024 ). [7]. ↵ Chouldechova A. Fair prediction with disparate impact: A study of bias in recidivism prediction instruments . arXiv [stat.AP] , http://arxiv.org/abs/1610.07524 ( 2016 ). [8]. ↵ Obermeyer Z , Powers B , Vogeli C , et al. Dissecting racial bias in an algorithm used to manage the health of populations . Science 2019 ; 366 : 447 – 453 . OpenUrl Abstract / FREE Full Text [9]. ↵ Vyas DA , Eisenstein LG , Jones DS . Hidden in Plain Sight — Reconsidering the Use of Race Correction in Clinical Algorithms . New England Journal of Medicine . Epub ahead of print 27 August 2020. DOI: 10.1056/nejmms2004740 . OpenUrl CrossRef PubMed [10]. ↵ Dwork C , Hardt M , Pitassi T , et al. Fairness through awareness . In: Proceedings of the 3rd Innovations in Theoretical Computer Science Conference . New York, NY, USA : ACM . Epub ahead of print 8 January 2012. DOI: 10.1145/2090236.2090255 . OpenUrl CrossRef [11]. ↵ Anderson JW , Visweswaran S. Algorithmic individual fairness and healthcare: a scoping review . JAMIA Open 2025 ; 8 : ooae149 . OpenUrl [12]. ↵ Fleisher W. What’s fair about individual fairness? In: Proceedings of the 2021 AAAI/ACM Conference on AI, Ethics, and Society . New York, NY, USA : ACM . Epub ahead of print 21 July 2021. DOI: 10.1145/3461702.3462621 . OpenUrl CrossRef [13]. ↵ Kusner MJ , Loftus JR , Russell C , et al. Counterfactual Fairness . Neural Inf Process Syst 2017 ; 30 : 4066 – 4076 . OpenUrl [14]. ↵ Subbaswamy A , Saria S. From development to deployment: dataset shift, causality, and shift-stable models in health AI . Biostatistics 2020 ; 21 : 345 – 352 . OpenUrl CrossRef PubMed [15]. ↵ Agniel D , Kohane IS , Weber GM . Biases in electronic health record data due to processes within the healthcare system: retrospective observational study . BMJ 2018 ; 361 : k1479 . OpenUrl Abstract / FREE Full Text [16]. ↵ Speicher T , Heidari H , Grgic-Hlaca N , et al. A unified approach to quantifying algorithmic unfairness: Measuring individual & group unfairness via inequality indices . arXiv [cs.LG] , http://arxiv.org/abs/1807.00787 ( 2018 ). [17]. ↵ Zhan D , Guo D , Ji P , et al. Bridging the fairness divide: Achieving group and individual fairness in graph neural networks . arXiv [cs.LG] , http://arxiv.org/abs/2404.17511 ( 2024 ). [18]. ↵ Wang Y , Pillai M , Zhao Y , et al. FairEHR-CLP: Towards fairness-aware Clinical Predictions with Contrastive Learning in multimodal electronic Health Records . arXiv [cs.LG] , http://arxiv.org/abs/2402.00955 ( 2024 , accessed 24 March 2025 ). [19]. ↵ Pillai M , Blumke TL , Studnia J , et al. Improving postsurgical fall detection for older Americans using LLM-driven analysis of clinical narratives . medRxiv 2024 ; 2024.06.25.24309480 . [20]. ↵ Hardt M , Price E , Srebro N. Equality of opportunity in supervised learning . arXiv [cs.LG] , http://arxiv.org/abs/1610.02413 ( 2016 , accessed 30 August 2025 ). [21]. ↵ Ofer D. COMPAS Recidivism Racial Bias , https://www.kaggle.com/danofer/compass ( 2017 , accessed 2 April 2026 ). [22]. ↵ Sagnik . UCI Adult Census Data Dataset , https://www.kaggle.com/sagnikpatra/uci-adult-census-data-dataset ( 2020 , accessed 2 April 2026 ). [23]. ↵ Chouldechova A. Fair prediction with disparate impact: A study of bias in recidivism prediction instruments . Big Data 2017 ; 5 : 153 – 163 . OpenUrl PubMed [24]. ↵ Angwin J , Larson J , Kirchner L , et al. Machine Bias — . ProPublica , https://www.propublica.org/article/machine-bias-risk-assessments-in-criminal-sentencing ( 2016 , accessed 29 April 2026 ). [25]. ↵ Ding F , Hardt M , Miller J , et al. Retiring Adult: New datasets for fair machine learning . arXiv [cs.LG] . Epub ahead of print 10 August 2021. DOI: 10.48550/arXiv.2108.04884 . OpenUrl CrossRef [26]. ↵ Pearl J. Causality , https://bayes.cs.ucla.edu/BOOK-2K/neuberg-review.pdf ( 2009 ). [27]. ↵ Loftus JR , Russell C , Kusner MJ , et al. Causal reasoning for algorithmic fairness . arXiv [cs.AI] , http://arxiv.org/abs/1805.05859 ( 2018 ). View the discussion thread. Back to top Previous Next Posted April 30, 2026. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Integrating Group and Individual Fairness Auditing in Clinical AI: A Post-Hoc, Model-Agnostic Approach Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Integrating Group and Individual Fairness Auditing in Clinical AI: A Post-Hoc, Model-Agnostic Approach Javen Xu , Yeon-Mi Hwang , Samhita Kondareddy , Inés Dormoy , Serena Liang Jing , Malvika Pillai , Catherine Curtin , Tina Hernandez-Boussard medRxiv 2025.09.03.25334999; doi: https://doi.org/10.1101/2025.09.03.25334999 Share This Article: Copy Citation Tools Integrating Group and Individual Fairness Auditing in Clinical AI: A Post-Hoc, Model-Agnostic Approach Javen Xu , Yeon-Mi Hwang , Samhita Kondareddy , Inés Dormoy , Serena Liang Jing , Malvika Pillai , Catherine Curtin , Tina Hernandez-Boussard medRxiv 2025.09.03.25334999; doi: https://doi.org/10.1101/2025.09.03.25334999 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15228) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6599) Geriatric Medicine (668) Health Economics (997) Health Informatics (4536) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (540) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9231) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a006c340fc3d09d6',t:'MTc3OTU2NzY0MQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00