Full text
39,040 characters
· extracted from
preprint-html
· click to expand
Automated Video-Based Analysis of Surgical Meta-competencies Using Computer Vision | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Automated Video-Based Analysis of Surgical Meta-competencies Using Computer Vision View ORCID Profile Josiah Aklilu , View ORCID Profile Joshua A. Villarreal , Chloe K. Nobuhara , View ORCID Profile Charlotte Egeland , Xiaohan Wang , View ORCID Profile Elaine Sui , View ORCID Profile Alan Brown , Matthew Leipzig , View ORCID Profile Reid Dale , View ORCID Profile Anita Rau , View ORCID Profile Alfred Song , Shelly Goel , View ORCID Profile Eric Sorenson , Vanessa Palter , Roger Bohn , View ORCID Profile Teodor Grantcharov , View ORCID Profile Jeffrey K. Jopling , View ORCID Profile Serena Yeung-Levy doi: https://doi.org/10.1101/2025.11.24.25340912 Josiah Aklilu 1 Department of Biomedical Data Science, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Josiah Aklilu Joshua A. Villarreal 2 Department of Surgery, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Joshua A. Villarreal Chloe K. Nobuhara 2 Department of Surgery, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Charlotte Egeland 3 Department of Transplantation and Digestive Diseases , Rigshospitalet, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Charlotte Egeland Xiaohan Wang 1 Department of Biomedical Data Science, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Elaine Sui 4 Department of Computer Science, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Elaine Sui Alan Brown 1 Department of Biomedical Data Science, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Alan Brown Matthew Leipzig 5 Department of Cardiothoracic Surgery, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Reid Dale 5 Department of Cardiothoracic Surgery, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Reid Dale Anita Rau 6 Intuitive Surgical Inc. , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Anita Rau Alfred Song 7 Department of Surgery, Intermountain Health , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Alfred Song Shelly Goel 8 NVIDIA Inc. , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Eric Sorenson 7 Department of Surgery, Intermountain Health , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Eric Sorenson Vanessa Palter 9 Surgical Safety Technologies Inc. , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Roger Bohn 10 School of Global Policy and Strategy, University of California San Diego , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Teodor Grantcharov 2 Department of Surgery, Stanford University , USA 9 Surgical Safety Technologies Inc. , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Teodor Grantcharov Jeffrey K. Jopling 11 Department of Surgery, Johns Hopkins School of Medicine , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jeffrey K. Jopling Serena Yeung-Levy 1 Department of Biomedical Data Science, Stanford University , USA 4 Department of Computer Science, Stanford University , USA 12 Clinical Excellence Research Center, Stanford University , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Serena Yeung-Levy For correspondence: syyeung{at}stanford.edu Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Background Traditional surgical training relies on an apprenticeship model, which is subjective and threatened by human bias. Performance metric scales attempt to offer more objective feedback by providing a structured grading rubric, but these scores are still ultimately subjective. Leveraging computer vision and artificial intelligence to assess surgical performance has the potential to shift subjective measurements into automated and objective feedback for trainees. Materials and Methods This retrospective, multi-institutional study analyzed 319 laparoscopic cholecystectomy videos from IRB-approved deidentified datasets and segmented the videos into 2862 clips. Using an internally validated video-based assessment rubric, we annotated video clips across five metacompetency domains: tissue handling, psychomotor skills, efficiency, dissection quality, and exposure quality. Short video segments ( < 90s) were rated on a 5-point scale by expert raters. We trained a deep learning model (DINOv2) to classify composite high (4–5) vs low (1–3) metacompetency scores, representing yes/no binary feedback, a realistic comparison to the operating room. Model performance was evaluated via area under the receiver operating characteristic curve. Results Among 2862 LC video clips, model performance was highest for dissection quality during the exposing gallbladder step (AUROC 91.5%, 95% confidence interval [CI], 84.5-96.5). Moderate performance was observed for efficiency (AUROC 72.6%, 95% CI 59.9-83.2) and exposure quality (AUROC 68.7%, 95% CI 55.2-81.8). Dissection and exposure quality scores during hepatocystic triangle dissection yielded AUROCs of 63.8% (95% CI 56.9-71.3) and 66.0% (95% CI 53.1-76.7), respectively. Conclusion We demonstrate the feasibility of a purely vision-based deep learning model to grade surgical skill based on metacompetencies with excellent performance during simple steps. This technique represents an advance over prior whole-video approaches that rely heavily on tool- tracking and kinematic data, and may lead to greater model performance by using binary feedback on increasingly specific step segmentation. Introduction The foundation of surgical education is built on a master-apprentice relationship between faculty and trainees ( Camison et al. 2022 ). This tried and true paradigm, while effective, has the potential to introduce subjective measurements and bias into surgical training ( Gerull et al. 2019 ; Vogt et al. 2003 ). Human bias in training can be favorable or unfavorable, and master surgeons can be influenced by their own surgical preferences in technique or the nature of the apprentice ( Vogt et al. 2003 ). To combat the subjective nature of training, several objective rubric-style measurements have been introduced in the past few decades. Objective Structured Assessment of Technical Skills (OSATS) and Entrustable Professional Activities (EPAs) represent two recent shifts towards measuring surgical competency in a more objective fashion ( Hiemstra et al. 2011 ; Hatala et al. 2015 ; Brasel et al. 2023 ). These grading scales can now provide numerical feedback, but are still influenced by a rater’s human bias ( Sherbino and Norman 2017 ). Artificial intelligence (AI) presents a unique opportunity to leverage the unbiased nature of computer programming to create truly objective and automated performance metrics ( Ryder et al. 2024 ; Igaki et al. 2023 ). Furthermore, AI is scalable; while the master-apprentice model requires oneto-one supervision, AI has the potential to automate assessment with immediate feedback to multiple individuals simultaneously. Previously, other groups have shown that AI can provide automated, objective feedback of surgical skills with comparable discrimination to a human reviewer ( Kiyasseh et al. 2023 ; Lavanchy et al. 2021 ). However, prior work has been limited in its performance by analyzing and scoring a surgical video based on a procedure in its entirety and relying heavily on tool-tracking and kinematic data ( Kiyasseh et al. 2023 ). We hypothesized that by segmenting a procedure into steps, we would be able to achieve higher agreement in human and model performance. Master surgeon raters would have higher levels of disagreement in more complex steps and higher levels of agreement (and therefore, better model performance) in simpler steps. We sought to address this question by taking a well-characterized cohort of 319 laparoscopic cholecystectomy (LC) videos and dividing them into 2862 video clips stratified by procedure step. We then trained a deep learning model to classify technical performance across five metacompetency (MC) domains: tissue handling, psychomotor skill, efficiency, dissection quality, and exposure quality ( Igaki et al. 2022 ). With these divisions in mind, the model was trained to judge each 45-70 second clip into a binary classifier of high or low performance, with high being an MC score of 4-5 and low being an MC score of 1-3, and we validated its performance with human ratings. This is, to our knowledge, the first study to show purely vision-based performance metrics of surgical metacompetencies in laparoscopic cholecystectomy. Materials and Methods Video Dataset Collection and Annotations This multi-institutional retrospective study included a dataset of 319 LC composed of de-identified publicly available and Institutional Review Board (IRB) approved intraoperative videos from two United States academic medical centers. We used an established surgical video annotation framework for video annotations and internally validated a video-based assessment (VBA) rubric for LC using validated universal surgical MC ( Meireles et al. 2021 ). A hierarchical task analysis for LC was conducted to define the intracorporeal steps for MC review. Ratings covered five skill domains, namely tissue handling, psychomotor skills, efficiency, dissection quality, and exposure quality. Meta-competency ratings were assigned on a 5-point Likert scale by a team of four expert raters ( Table 1 ). View this table: View inline View popup Download powerpoint Table 1: Characterization of video clip dataset. We report the total number of video clips from our mutli-source dataset and the number of unique procedures the clips were extracted from. View this table: View inline View popup Download powerpoint Table 2: Surgical meta-competency annotations. Total counts of video clips from each step of laparoscopic cholecystectomy, and training and test splits for model development. We use a 60:20:20 split for all competency models. Counts in the meta-competency columns represent the number of high vs low -skill samples (respectively) included per procedural step on the test set. Model Architecture and Training Our model architecture consisted of a DINOv2 ( Oquab et al. 2024 ), a pretrained vision transformer backbone (ViT-B) and a multi-stage temporal convolutional network (TCN) with 4 TCN layers ( Lea et al. 2016 ). DINOv2 is an advanced artificial intelligence (AI) system for image analysis. It learns to identify important visual features and patterns by analyzing millions of general, nonmedical images on its own, without requiring manual labeling by human experts. We leveraged DINOv2, which was trained on a corpus of 142 million web-crawled images, to produce highlydescriptive features of each video frame in the surgical clips. The TCN is another type of computer vision-based AI system that learns temporal relationships. We used TCN to process the DINOv2 features extracted from each frame, allowing the model to learn temporal relationships between video frame features throughout a single clip. Thus, each meta-competency model comprised of a TCN over video frame features extracted by DINOv2 for each clip. To train these models, we used Low-Rank Adaptation (LoRA) [ Hu et al. 2021 ] fine-tuning for the DINOv2 vision backbone and full parameter fine-tuning of the TCN head. LoRA is a computationally efficient method to specialize large AI models for new tasks. Instead of retraining the entire model, LoRA freezes the original model’s knowledge and trains only a small set of new parameters. This approach significantly reduces the computational resources required and, crucially, prevents the loss of the original model’s powerful foundational knowledge when adapting it to a specialized dataset. In our study, we leveraged the powerful, general-purpose representations from the DINOv2 model while specializing it for our surgical video dataset, which has a distinct visual domain. Also for training, we used the AdamW optimizer for adaptively updating our model parameters with each training step. This optimizer allows the model to adjust individual parameters at different rates to adjust its outputs for each batch of video clips it sees during training. We use a learning rate of 1 e −6 for the backbone and 3 e −5 for the TCN head ( Figure 1 ). We also implemented weight decay of 0.01 for the backbone and 0.05 for the TCN head, along with dropout rate of 0.3 for regularization. These two techniques prevent the model from memorizing training videos clips and help with generalization to new videos. We trained our models for 20 epochs on our training datasets. The models were then trained to differentiate between a binarized composite of low (1-3) and high (4-5) levels of operative performance according to MC ratings. Download figure Open in new tab Figure 1: Computer Vision Model Architecture. Design of the computer vision model used to classify adequacy in each surgical meta-competency. Results Our dataset used a total of 319 intraoperative, deidentified videos of LC from two United States academic medical centers (Proprietary dataset 1, Proprietary dataset 2) and international surgical collaborators (Cholec80, Heichole) [ Table 1 ]. The initial stage of clip generation involved first segmenting each video into four steps of the LC procedure: (1) Expose gallbladder, (2) Dissection of hepatocystic triangle, (3) Ligation and division, (4) Gallbladder removal. Segments longer than 360 seconds were divided into four quartiles, with one 60–90 second clip drawn from each. A maximum clip duration of 90-seconds was chosen as the high end value as an interval representing the practical timeframe for video annotators to rate a single surgical maneuver. This clip generation procedure created 2891 clips, of which 2862 were usable for annotations from human reviewers (e.g. 29 clips were excluded for lack of active operating, poor image quality, or extracorporeal footage) [ Table 1 ]. All 2862 clips received annotations from human reviewers on the five metacompetency (MC) domains: tissue handling, psychomotor skill, efficiency, dissection quality, and exposure quality. A subset of dataset was labeled by multiple reviewers, with a planned overlap in assignments to allow for the measurement of inter-rater reliability. We reported model discrimination performance with area under the receiver operating characteristic curve (AUROC) in our testing data. We employed AUROC as our primary evaluation metric because it offers a robust measure of the classifier’s performance across a range of decision thresholds. Additionally, we used 1000 bootstrap samples to compute all 95% confidence intervals in our analysis. Our model achieved its highest performance for the exposing gallbladder (EG) step, achieving an AUROC of 91.5% (95% confidence interval [CI], 84.5-96.5) for discriminating dissection quality ( Figure 2 ). Additionally, for the EG step, the model demonstrated modest performance for efficiency (AUROC 72.6%, 95% CI, 59.9-83.2) and exposure equality (AUROC 68.7%, 95% CI, 55.2-81.8). For the dissection of hepatocystic triangle (HCT) step where identification of the critical view of safety is achieved, our model demonstrated an AUROC of 66.0% (95% CI, 53.1-76.7) in classifying exposure quality and 63.8% (95% CI, 56.9-71.3) for dissection quality. Download figure Open in new tab Figure 2: Model Discrimination Metrics for Meta-competency Predictions Across Cholecystectomy Steps. We plot the performance for each of our meta-competency models per intracoproreal step. Note that the Ligation & Division step does not have ratings for dissection quality since no dissection occurs during this step. Error bars indicate 95% confidence intervals. Abbreviations: HCT, hepatocystic triangle. Discussion Surgical pedagogy has traditionally relied on a one-to-one training and expert opinion, which is subjective, threatened by human biases, and limited in its scalability. In the age of artificial intelligence (AI), there is a unique opportunity to capitalize on the unbiased and calculated nature of machine learning to transfer performance metrics from what is currently a subjective art into an objective science. Here, we tested the feasibility of using a pure vision-only approach to classify laparoscopic cholecystectomy (LC) into high versus low skill, by segmenting the LC procedure into key steps, and found excellent model agreement in simpler steps. More complex steps may need further subdivision to achieve this level of agreement between AI and human raters. Currently, AI has been used in other proof-of-concept studies to automate surgical performance metrics. These studies either use kinematics (e.g. physical data from or either motion sensors, force feedback, or three-dimensional coordinates from robotics) or computer vision (e.g. using programmed algorithms to understand videos in real time and extract data) [ Kiyasseh et al. 2023 ; Soangra et al. 2022 ]. Following data collection, there are several performance metrics where automated analysis can be completed using machine learning. Deep learning, a subset of machine learning, utilizes multi-layered architectures inspired by the human brain (i.e. neural networks) to learn increasingly complex features from high-dimensional video ( Pedrett et al. 2023 ). We used DINOv2, a deep learning algorithm trained on a curated dataset of > 140 million images and that excels in tasks requiring temporal/spatial understanding to produce descriptive visual features for each surgical video clip ( Oquab et al. 2024 ). Other studies have primarily focused on used machine learning and a combination of kinematics or tool tracking data to quantify performance metrics ( Khalid et al. 2020 ; Kiyasseh et al. 2023 ; Lam et al. 2022 ; Lavanchy et al. 2021 ). Few studies have used a ordered scale of proficiency to train a model to recognize level of skill based on a metacompetency (MC)-based metrics ( Pérez-Escamirosa et al. 2020 ; Pan et al. 2023 ). No study to date has then segmented the procedure into clips to then create a binary readout of high skill (MC 4-5) or low skill (MC 1-3). We initially hypothesized that variations in inter-rater reliability across the entire procedure were high because the operator can perform with both high skill and low skill in the same operation, and raters may anchor to one demonstration more than the other. Therefore, by segmenting the procedure into clips, we would achieve better model performance and excellent performance in the simpler steps (e.g. rating the dissection quality during gallbladder exposure). While our model demonstrated the feasibility of using computer vision to classify surgical skill with excellent accuracy, only one model achieved excellent performance of over 90% for dissection quality of exposing the gallbladder. Other MC metrics in the EG step achieved modest performance of approximately 70%. Notably, and arguably where evaluating performance matters the most, the dissection of the hepatocystic triangle step reached at highest an AUROC of 66% despite this step being the largest clip dataset at n=950. This indicates that human raters still have high levels of disagreement in assessing objective performance metrics at complex and critical steps in the operation. Furthermore, we used our own internal video-based assessment rubric. External validation of video-based assessments in MC domains (e.g. via a modified Delphi process) can promote standardization in AI assessments ( Meireles et al. 2021 ). Model development in the era of computer vision and surgical performance is limited by variations in human labels and degree of disagreement. Two surgeons watching a 60-minute clip are likely to grade the overall MCs differently due to internal biases and mental anchoring on a high skill or low skill portion of performance during the procedure. The clip generation portion of this study is a novel approach to using computer vision to extract data from surgical videos, and combining this with a binary performance metric of high versus low skill mirrors the yes/no feedback that a trainee is more likely to hear while operating. From a computer vision standpoint, this approach represents an entirely new workflow and paradigm for facilitating human rater disagreement by segmenting longer videos into shorter segments. From a surgical standpoint, this study represents a proof-ofconcept approach to using computer vision and surgical video clips to automate objective feedback and assessment for simple steps of an operation. Data Availability All data produced in the present study are available upon reasonable request to the authors. Conflicts of Interest Disclosure Teodor Grantcharov MD, PhD is founder of Surgical Safety Technologies, Vanessa Palter MD, PhD is an employee of Surgical Safety Technologies, Joshua A. Villarreal and Chloe K. Nobuhara are scientific advisors for Surgical Safety Technologies. Funding This work was supported by Wellcome Leap SAVE (No. 63447087-287892) and partially supported by the Stanford Clinical Excellence Research Center. The funders had no role in the design and conduct of the study; collection, management, analysis, and interpretation of the data; preparation, review, or approval of the manuscript; and decision to submit the manuscript for publication. Footnotes ↵ * co-first authors ↵ ** co-senior authors References ↵ Brasel , Karen J. , Brenessa Lindeman , Andrew Jones , George A. Sarosi , Rebecca Minter , Mary E. Klingensmith , James Whiting , David Borgstrom , Jo Buyske , and John D. Mellinger . 2023 . “Implementation of Entrustable Professional Activities in General Surgery: Results of a National Pilot Study” [in eng] . Annals of Surgery 278 , no. 4 ( October ): 578 – 586 . ISSN: 1528-1140 . doi: 10.1097/SLA.0000000000005991 . OpenUrl CrossRef PubMed ↵ Camison , Liliana , Jack E. Brooker , Sanjay Naran , John R. Potts , and Joseph E. Losee . 2022 . “ The History of Surgical Education in the United States: Past, Present, and Future .” Annals of Surgery Open 3 , no. 1 ( March ): e148 . ISSN: 2691-3593 , accessed November 22, 2025 . doi: 10.1097/AS9.0000000000000148 . https://pmc.ncbi.nlm.nih.gov/articles/PMC10013151/ . OpenUrl CrossRef ↵ Gerull , Katherine M. , Maren Loe , Kristen Seiler , Jared McAllister , and Arghavan Salles . 2019 . “Assessing gender bias in qualitative evaluations of surgical residents” [in English] . Publisher: Elsevier , The American Journal of Surgery 217 , no. 2 ( February ): 306 – 313 . ISSN: 0002-9610, 1879-1883 , accessed November 22, 2025 . doi: 10.1016/j.amjsurg.2018.09.029 . https://www.americanjournalofsurgery.com/article/S0002-9610(18)30631-7/abstract . OpenUrl CrossRef PubMed ↵ Hatala , Rose , David A. Cook , Ryan Brydges , and Richard Hawkins . 2015 . “Constructing a validity argument for the Objective Structured Assessment of Technical Skills (OSATS): a systematic review of validity evidence” [in en] . Advances in Health Sciences Education 20 , no. 5 ( December ): 1149 – 1175 . ISSN: 1573-1677 , accessed November 22, 2025 . doi: 10.1007/s10459-015-9593-1 . https://doi.org/10.1007/s10459-015-9593-1 . OpenUrl CrossRef PubMed ↵ Hiemstra , Ellen , Wendela Kolkman , Ron Wolterbeek , Baptist Trimbos , and Frank Willem Jansen . 2011 . “ Value of an objective assessment tool in the operating room .” Canadian Journal of Surgery 54 , no. 2 ( April ): 116 – 122 . ISSN: 0008-428X , accessed November 22, 2025 . doi: 10.1503/cjs.032909 . https://pmc.ncbi.nlm.nih.gov/articles/PMC3116698/ . OpenUrl Abstract / FREE Full Text ↵ Hu , Edward J. , Yelong Shen , Phillip Wallis , Zeyuan Allen-Zhu , Yuanzhi Li , Shean Wang , Lu Wang , and Weizhu Chen . 2021 . LoRA: Low-Rank Adaptation of Large Language Models . arXiv: 2106.09685 [cs.CL]. https://arxiv.org/abs/2106.09685 . ↵ Igaki , Takahiro , Daichi Kitaguchi , Hiroki Matsuzaki , Kei Nakajima , Shigehiro Kojima , Hiro Hasegawa , Nobuyoshi Takeshita , Yusuke Kinugasa , and Masaaki Ito . 2023 . “ Automatic Surgical Skill Assessment System based on concordance of Standardized Surgical Field Development using artificial intelligence .” JAMA Surgery 158 , no. 8 (August). doi: 10.1001/jamasurg.2023.1131 . OpenUrl CrossRef ↵ Igaki , Takahiro , Shin Takenaka , Yusuke Watanabe , Shigehiro Kojima , Kei Nakajima , Yuya Takabe , Daichi Kitaguchi , et al. 2022 . “ Universal meta-competencies of operative performances: A literature review and qualitative synthesis .” Surgical Endoscopy 37 , no. 2 ( September ): 835 – 845 . doi: 10.1007/s00464-022-09573-4 . OpenUrl CrossRef PubMed ↵ Khalid , Shuja , Mitchell Goldenberg , Teodor Grantcharov , Babak Taati , and Frank Rudzicz . 2020 . “ Evaluation of Deep Learning Models for Identifying Surgical Actions and Measuring Performance .” JAMA Network Open 3 , no. 3 ( March ): e201664 . ISSN: 2574-3805 , accessed November 22, 2025 . doi: 10.1001/jamanetworkopen.2020.1664 . https://doi.org/10.1001/jamanetworkopen.2020.1664 . OpenUrl CrossRef ↵ Kiyasseh , Dani , Runzhuo Ma , Taseen F. Haque , Brian J. Miles , Christian Wagner , Daniel A. Donoho , Animashree Anandkumar , and Andrew J. Hung . 2023 . “A vision transformer for decoding surgeon activity from surgical videos” [in en] . Publisher: Nature Publishing Group, Nature Biomedical Engineering 7 , no. 6 ( June ): 780 – 796 . ISSN: 2157-846X , accessed November 22, 2025 . doi: 10.1038/s41551-023-01010-8 . https://www.nature.com/articles/s41551-023-01010-8 . OpenUrl CrossRef ↵ Lam , Kyle , Frank P.-W. Lo , Yujian An , Ara Darzi , James M. Kinross , Sanjay Purkayastha , and Benny Lo . 2022 . “ Deep Learning for Instrument Detection and Assessment of Operative Skill in Surgical Videos .” IEEE Transactions on Medical Robotics and Bionics 4 , no. 4 ( November ): 1068 – 1071 . ISSN: 2576-3202 , accessed November 22, 2025 . doi: 10.1109/TMRB.2022.3214377 . https://ieeexplore.ieee.org/document/9917447 . OpenUrl CrossRef ↵ Lavanchy Joël L. , Joel Zindel , Kadir Kirtac , Isabell Twick , Enes Hosgor , Daniel Candinas , and Guido Beldi . 2021 . “Automation of surgical skill assessment using a three-stage machine learning algorithm” [in en] . Publisher: Nature Publishing Group , Scientific Reports 11 , no. 1 ( March ): 5197 . ISSN: 2045-2322 , accessed November 22, 2025 . doi: 10.1038/s41598-021-84295-6 . https://www.nature.com/articles/s41598-021-84295-6 . OpenUrl CrossRef PubMed ↵ Lea , Colin , Rene Vidal , Austin Reiter , and Gregory D. Hager . 2016 . Temporal Convolutional Networks: A Unified Approach to Action Segmentation . arxiv: 1608.08242 [cs.CV]. https://arxiv.org/abs/1608.08242 . ↵ Meireles , Ozanan R. , Guy Rosman , Maria S. Altieri , Lawrence Carin , Gregory Hager , Amin Madani , Nicolas Padoy , et al. 2021 . “ Sages consensus recommendations on an annotation framework for Surgical Video .” Surgical Endoscopy 35 , no. 9 ( July ): 4918 – 4929 . doi: 10.1007/s00464-021-08578-9 . OpenUrl CrossRef PubMed ↵ Oquab , Maxime , Timothée Darcet , Théo Moutakanni , Huy Vo , Marc Szafraniec , Vasil Khalidov , Pierre Fernandez , et al. 2024 . DINOv2: Learning Robust Visual Features without Supervision . arxiv: 2304.07193 [cs.CV]. https://arxiv.org/abs/2304.07193 . ↵ Pan , Mingzhang , Shuo Wang , Jingao Li , Jing Li , Xiuze Yang , and Ke Liang . 2023 . “An Automated Skill Assessment Framework Based on Visual Motion Signals and a Deep Neural Network in Robot-Assisted Minimally Invasive Surgery” [in en] . Publisher: Multidisciplinary Digital Publishing Institute , Sensors 23 , no. 9 ( January ): 4496 . ISSN: 1424-8220 , accessed November 22, 2025 . doi: 10.3390/s23094496 . https://www.mdpi.com/1424-8220/23/9/4496 . OpenUrl CrossRef PubMed ↵ Pedrett , Romina , Pietro Mascagni , Guido Beldi , Nicolas Padoy , and Joël L. Lavanchy . 2023 . “ Technical skill assessment in minimally invasive surgery using artificial intelligence: a systematic review .” Surgical Endoscopy 37 ( 10 ): 7412 – 7424 . ISSN: 0930-2794 , accessed November 22, 2025 . doi: 10.1007/s00464-023-10335-z . https://pmc.ncbi.nlm.nih.gov/articles/PMC10520175/ . OpenUrl CrossRef PubMed ↵ Pérez-Escamirosa , Fernando , Antonio Alarcón-Paredes , Gustavo Adolfo Alonso-Silverio , Ignacio Oropesa , Oscar Camacho-Nieto , Daniel Lorias-Espinoza , and Arturo Minor-Martínez . 2020 . “Objective classification of psychomotor laparoscopic skills of surgeons based on three different approaches” [in en] . International Journal of Computer Assisted Radiology and Surgery 15 , no. 1 ( January ): 27 – 40 . ISSN: 1861-6429 , accessed November 22, 2025 . https://doi.org/10.1007/s11548-019-02073-2 . doi: 10.1007/s11548-019-02073-2 . OpenUrl CrossRef PubMed ↵ Ryder , C. Yoonhee , Nicole M. Mott , Christopher L. Gross , Chioma Anidi , Leul Shigut , Serena S. Bidwell , Erin Kim , et al. 2024 . “ Using artificial intelligence to gauge competency on a novel Laparoscopic Training System .” Journal of Surgical Education 81 , no. 2 ( February ): 267 – 274 . doi: 10.1016/j.jsurg.2023.10.007 . OpenUrl CrossRef PubMed ↵ Sherbino , Jonathan , and Geoff Norman . 2017 . “ On Rating Angels: The Halo Effect and Straight Line Scoring .” Journal of Graduate Medical Education 9 , no. 6 ( December ): 721 – 723 . ISSN: 1949-8349 , accessed November 22, 2025 . doi: 10.4300/JGME-D-17-00644.1 . https://pmc.ncbi.nlm.nih.gov/articles/PMC5734326/ . OpenUrl CrossRef PubMed ↵ Soangra , Rahul , R. Sivakumar , E. R. Anirudh , Sai Viswanth Reddy Y , and Emmanuel B. John . 2022 . “Evaluation of surgical skill using machine learning with optimal wearable sensor locations” [in en] . Publisher: Public Library of Science , PLOS ONE 17 , no. 6 ( June ): e0267936 . ISSN: 1932-6203 , accessed November 22, 2025 . doi: 10.1371/journal.pone.0267936 . https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0267936 . OpenUrl CrossRef PubMed ↵ Vogt , Val Y. , Vanessa M. Givens , Craig A. Keathley , Gary H. Lipscomb , and Robert L. Summitt . 2003 . “Is a resident’s score on a videotaped objective structured assessment of technical skills affected by revealing the resident’s identity?” [In English] . Publisher: Elsevier , American Journal of Obstetrics & Gynecology 189 , no. 3 ( September ): 688 – 691 . ISSN: 0002-9378, 1097-6868 , accessed November 22, 2025 . doi: 10.1067/S0002-9378(03)00887-1 . https://www.ajog.org/article/S0002-9378(03)00887-1/abstract . OpenUrl CrossRef PubMed Web of Science View the discussion thread. Back to top Previous Next Posted November 27, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Automated Video-Based Analysis of Surgical Meta-competencies Using Computer Vision Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Automated Video-Based Analysis of Surgical Meta-competencies Using Computer Vision Josiah Aklilu , Joshua A. Villarreal , Chloe K. Nobuhara , Charlotte Egeland , Xiaohan Wang , Elaine Sui , Alan Brown , Matthew Leipzig , Reid Dale , Anita Rau , Alfred Song , Shelly Goel , Eric Sorenson , Vanessa Palter , Roger Bohn , Teodor Grantcharov , Jeffrey K. Jopling , Serena Yeung-Levy medRxiv 2025.11.24.25340912; doi: https://doi.org/10.1101/2025.11.24.25340912 Share This Article: Copy Citation Tools Automated Video-Based Analysis of Surgical Meta-competencies Using Computer Vision Josiah Aklilu , Joshua A. Villarreal , Chloe K. Nobuhara , Charlotte Egeland , Xiaohan Wang , Elaine Sui , Alan Brown , Matthew Leipzig , Reid Dale , Anita Rau , Alfred Song , Shelly Goel , Eric Sorenson , Vanessa Palter , Roger Bohn , Teodor Grantcharov , Jeffrey K. Jopling , Serena Yeung-Levy medRxiv 2025.11.24.25340912; doi: https://doi.org/10.1101/2025.11.24.25340912 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Surgery Subject Areas All Articles Addiction Medicine (570) Allergy and Immunology (864) Anesthesia (302) Cardiovascular Medicine (4445) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1515) Epidemiology (15236) Forensic Medicine (30) Gastroenterology (1127) Genetic and Genomic Medicine (6610) Geriatric Medicine (669) Health Economics (1000) Health Informatics (4549) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15926) Intensive Care and Critical Care Medicine (1104) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6613) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1147) Occupational and Environmental Health (957) Oncology (3341) Ophthalmology (975) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1694) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5458) Public and Global Health (9244) Radiology and Imaging (2205) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1197) Rheumatology (596) Sexual and Reproductive Health (715) Sports Medicine (530) Surgery (713) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a0234ac64df08650',t:'MTc3OTg2Njc5Mw=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.