@article{Jenkins:2026aa,
 abstract = {OBJECTIVE
To evaluate resident versus attending operative notes using a two-phase approach combining natural language processing (NLP) diffing and structured assessment for evaluating operative reports (SAFE-OR) structured grading, with grader-severity adjustment to account for inter-rater differences. Operative notes document procedural conduct, support continuity of care, and have educational value. While prior studies show resident-authored notes often contain omissions or errors, their potential as a competency assessment tool remains untapped due to the difficulty of analyzing free text at scale. NLP and validated structured assessment tools offer new opportunities to objectively measure the quality and content of these notes.
DESIGN, SETTING, AND PARTICIPANTS
We extracted operative notes for common general surgery procedures (2010-2023) that had both a resident-authored and an attending-signed version for the same encounter. In Phase 1, an NLP-based ``diffing'' pipeline quantified insertions, deletions, and replacements between paired notes. In Phase 2, ten blinded graders (attendings and senior residents) evaluated de-identified notes with the SAFE-OR checklist and quality scale. To correct for inter-grader-severity differences, scores were z-normalized within each grader and rescaled to SAFE-OR ranges before resident versus attending comparisons. Paired t-tests, Wilcoxon signed-rank tests, and mixed-effects models were used for hypothesis testing; procedure-specific analyses were performed for common operations.
RESULTS
A total of 2949 notes were analyzed with NLP diffing, revealing significant variation in edit burden across procedures (p < 0.001), with esophagogastroduodenoscopy with dilation showing the highest change rate. SAFE-OR grading included 139,439 grader-level scores across 75 matched note pairs. After grader-severity adjustment, resident-authored notes scored significantly lower than attending-authored notes in Section I (preoperative elements, mean diff −0.35, p = 0.005), Section III (postoperative/disposition, mean diff −1.70, p = 0.022), and overall score (mean diff −2.31, p = 0.029). Section II (intraoperative elements) differences were not statistically significant. Mixed-effects models confirmed these findings (overall coef −0.039, p < 0.001). Procedure-specific analysis showed significant differences for laparoscopic appendectomy but not for laparoscopic cholecystectomy or laparoscopic cholecystectomy with cholangiogram.
CONCLUSIONS
Resident operative notes contained fewer key elements and scored lower in structured quality assessments compared with attending notes, particularly in preoperative and postoperative documentation. Combining NLP-based edit quantification with validated grading instruments enables high-throughput, objective assessment of operative note quality and offers a novel pathway for evaluating surgical trainees' cognitive understanding of operations. Future work will apply large language models for thematic analysis to identify the nature of content differences.},
 author = {Phillip D. Jenkins and Eric Cramer and Julie Doberne and Steven Bedrick and Kenneth Azarow and Ruchi Thanawala},
 bdsk-url-1 = {https://www.sciencedirect.com/science/article/pii/S1931720426001546},
 bdsk-url-2 = {https://doi.org/10.1016/j.jsurg.2026.104022},
 date-added = {2026-06-16 08:56:22 -0700},
 date-modified = {2026-06-25 09:55:03 -0700},
 doi = {https://doi.org/10.1016/j.jsurg.2026.104022},
 issn = {1931-7204},
 journal = {Journal of Surgical Education},
 keywords = {operative notes, surgical education, natural language processing, competency-based assessment, resident documentation},
 month = {September},
 number = {9},
 pages = {104022},
 title = {Leveraging Natural Language Processing and Structured Scoring to Evaluate Operative Note Quality in Surgical Training},
 url = {https://www.sciencedirect.com/science/article/pii/S1931720426001546},
 volume = {83},
 year = {2026}
}
