<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id><journal-id journal-id-type="publisher-id">mededu</journal-id><journal-id journal-id-type="index">20</journal-id><journal-title>JMIR Medical Education</journal-title><abbrev-journal-title>JMIR Med Educ</abbrev-journal-title><issn pub-type="epub">2369-3762</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v12i1e95578</article-id><article-id pub-id-type="doi">10.2196/95578</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Evaluating Large Language Model&#x2013;Based Automated Scoring in a Voice-Based Virtual Standardized Patient Platform for Medical Students: Cross-Sectional Agreement Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Gao</surname><given-names>Xiaoxing</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Huang</surname><given-names>Xiaoming</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hu</surname><given-names>Rongrong</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Li</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Huiting</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Bingqing</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wei</surname><given-names>Chong</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Qiu</surname><given-names>Wei</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Mengyu</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Sun</surname><given-names>Xuefeng</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff9">9</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Pulmonary and Critical Care Medicine, Peking Union Medical College Hospital</institution><addr-line>1 Shuaifuyuan, Dongcheng District</addr-line><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of General Internal Medicine, Peking Union Medical College Hospital</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Nephrology, Peking Union Medical College Hospital</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff4"><institution>Department of Rheumatology, Peking Union Medical College Hospital</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff5"><institution>Department of Infectious Disease, Peking Union Medical College Hospital</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff6"><institution>Department of Haematology, Peking Union Medical College Hospital</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff7"><institution>Department of Medical Oncology, Peking Union Medical College Hospital</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff8"><institution>Department of Neurology, Peking Union Medical College Hospital</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff9"><institution>Department of Internal Medicine, Peking Union Medical College Hospital</institution><addr-line>Beijing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Lee</surname><given-names>Jason Wen Yau</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Vicente</surname><given-names>Maria Asuncion</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Cui</surname><given-names>Shasha</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Xuefeng Sun, MD, Department of Pulmonary and Critical Care Medicine, Peking Union Medical College Hospital, 1 Shuaifuyuan, Dongcheng District, Beijing, 100730, China, 86 10 6915 5000; <email>sunxfer@sina.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>24</day><month>9</month><year>2026</year></pub-date><volume>12</volume><elocation-id>e95578</elocation-id><history><date date-type="received"><day>19</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>28</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Xiaoxing Gao, Xiaoming Huang, Rongrong Hu, Li Zhang, Huiting Liu, Bingqing Zhang, Chong Wei, Wei Qiu, Mengyu Zhang, Xuefeng Sun. Originally published in JMIR Medical Education (<ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org">https://mededu.jmir.org</ext-link>), 24.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org/">https://mededu.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mededu.jmir.org/2026/1/e95578"/><abstract><sec><title>Background</title><p>Large language model (LLM)&#x2013;powered virtual standardized patients (VSPs) enable scalable clinical skills practice, but the validity of AI-generated scores relative to faculty ratings remains unclear.</p></sec><sec><title>Objective</title><p>This study aimed to assess agreement between LLM-generated and faculty ratings of history-taking and communication performance and to examine the influence of rater and case heterogeneity.</p></sec><sec sec-type="methods"><title>Methods</title><p>In this cross-sectional study, 92 fourth-year medical students completed one of three 15-minute voice-based VSP cases (fever, diarrhea, and cough). Ten blinded faculty raters scored performance (0&#x2010;100 points total; 0&#x2010;50 points per domain). AI scores were generated by DeepSeek-V3 using a calibrated prompt. Agreement was evaluated using mixed-effects models, intraclass correlation coefficients (ICC [2,1]), Spearman correlations, mean absolute error (MAE), Bland-Altman analysis, and variance partition coefficients (VPC).</p></sec><sec sec-type="results"><title>Results</title><p>Median total scores were similar for AI and faculty (median 93.0, IQR 89.0-95.0 vs median 94.0, IQR 91.0-95.0). Rater variability accounted for 37% of residual variance in faculty total scores (VPC=0.37). AI total scores were positively associated with faculty total scores (&#x03B2;=0.37, 95% CI 0.26&#x2010;0.48; <italic>P</italic>&#x003C;.001; Spearman &#x03C1;=0.50, 95% CI 0.34&#x2010;0.65). Absolute agreement was moderate (ICC[2,1]=0.51, 95% CI 0.34&#x2010;0.65), with MAE of 3.11 points. Mixed-effects Bland-Altman analysis showed a small, not statistically significant mean bias (1.26 points, 95% CI &#x2212;0.48 to 3.01; <italic>P</italic>=.16) and 95% limits of agreement from &#x2212;4.95 to 7.48 (width=12.43 points), with proportional bias (&#x03B2;_proportional bias=&#x2212;0.55; <italic>P</italic>&#x003C;.001). Agreement was stronger for information gathering (&#x03B2;=0.46; &#x03C1;=0.49; ICC=0.54; VPC=0.23) than for communication (&#x03B2;=0.27; &#x03C1;=0.28; ICC=0.29; VPC=0.52). A sensitivity analysis in the lowest quartile showed attenuated but consistent agreement (ICC=0.38).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>LLM-based scoring in a VSP showed moderate agreement with faculty ratings, performing better for information gathering than for communication. Due to rater and case heterogeneity, ceiling effects, and proportional bias, this method is suitable for formative use and enhanced sampling in programmatic assessment but not for independent, high-stakes summative decisions.</p></sec></abstract><kwd-group><kwd>large language model</kwd><kwd>virtual standardized patient</kwd><kwd>information gathering</kwd><kwd>communication skills</kwd><kwd>rater variability</kwd><kwd>mixed-effects model</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>History-taking and physician-patient communication are core competencies in undergraduate medical education. Traditional teaching relies on faculty-facilitated role-play and standardized patients (SPs), which are resource-intensive and constrained by faculty time and SP availability. As a result, students often have limited opportunities for repeated practice and timely feedback.</p><p>Virtual standardized patients (VSPs) are computer-based simulations designed to portray patients with defined clinical presentations and respond dynamically to learner inquiries. They offer a scalable supplement or alternative. Virtual patients have long been used to support history-taking and clinical reasoning, and reviews show positive learner perceptions and modest gains in knowledge and skills compared with traditional methods [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Earlier systems largely used branching logic and rule-based scoring. Recent large language models (LLMs) can support free-text, contextually appropriate dialogue, enabling more natural and flexible patient simulations [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>Beyond dialogue, LLMs can generate automated scores and feedback, potentially increasing feedback frequency while reducing faculty workload [<xref ref-type="bibr" rid="ref5">5</xref>]. However, the usefulness of automated assessment depends on validity and reliability. For summative uses (eg, progression decisions), strong agreement with expert judgment and robust validity evidence are essential; for formative use, moderate agreement may be acceptable if feedback is timely, specific, and actionable [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Notably, even trained human raters in objective structured clinical examinations (OSCEs) and workplace-based assessments often show only moderate interrater reliability and substantial variability in severity and leniency [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Therefore, AI-human agreement should be interpreted in the context of human-human variability, rather than assuming a flawless human &#x201C;gold standard.&#x201D; Few empirical studies have systematically compared LLM-driven scoring of clinical communication with human examiners in authentic teaching settings&#x2014;especially within integrated VSP platforms [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>We developed an LLM-based VSP system for voice-based history-taking and communication practice. Students conducted spoken interviews with an LLM-driven virtual patient; the system then generated scores and narrative feedback. We evaluated the system by comparing LLM-generated scores with faculty ratings across multiple instructional groups and clinical cases.</p><p>This study aimed to evaluate the agreement between LLM-generated and faculty ratings of history-taking and communication performance in an LLM-powered VSP platform and to examine how rater and case heterogeneity influenced agreement to determine appropriate educational use boundaries for LLM-based scoring.</p></sec><sec id="s1-2"><title>Research Questions</title><p>This study addressed three questions: (1) How closely do LLM total scores agree with faculty ratings? (2) How do rater and case factors affect AI-human agreement? (3) Is the psychometric agreement between LLM-based scoring and faculty ratings sufficient to support formative and/or summative assessment uses?</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Setting</title><p>This cross-sectional observational study was conducted at Peking Union Medical College Hospital. Each student completed a single VSP encounter via a voice-based dialogue. In each encounter, the VSP acted as the patient, while students role-played the physician to conduct a history-taking interview. Encounters were limited to 15 minutes. Immediately after the encounter, the supervising faculty member provided ratings while blinded to AI-generated scores.</p></sec><sec id="s2-2"><title>Participants</title><p>Ninety-two fourth-year medical students enrolled in a diagnostics course participated. Prior to the experiment, all participants had completed 7 weeks of history-taking training, including 1 human SP practice session per week, and they were offered voluntary access to an LLM-powered VSP platform aligned with the SP curriculum. Students could initiate unlimited, untimed multiturn interviews with any case, at any time, outside scheduled SP sessions.</p><p>As preparation for the end-of-term history-taking examination, we conducted this VSP-based exercise. Students were informed that this was an AI-simulated practice session; they were not told that AI scores would be compared with faculty ratings for research. Ten faculty raters scored students using a standardized rubric, with brief orientation but no formal calibration.</p></sec><sec id="s2-3"><title>LLM-Powered VSP System</title><p>The VSP system used DeepSeek-V3 (DeepSeek AI, accessed via API, temperature=0.0) for scoring, Doubao-1.5 (ByteDance) for patient dialogue, paraformer-realtime-v2 (Alibaba) for automatic speech recognition (ASR), and CosyVoice-v2 (Alibaba) for text-to-speech synthesis. Prior to deployment, we iteratively calibrated the LLM scoring prompt using 45 historical records from the 2024 academic year (independent cohort, no overlap in students, cases, or raters). The five-stage calibration involved (1) rubric-aligned prompt drafting, (2) initial testing on 15 records, (3) item-level discrepancy analysis, (4) iterative refinement targeting systematic errors, and (5) validation on the held-out 30 records. The prompt was frozen before deployment. The complete prompt and detailed calibration procedure are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>Three clinical cases (fever, diarrhea, and cough) were developed by the same clinical education team to represent common outpatient presentations of comparable complexity and expected duration (15 min). Participants were randomly assigned to one of 3 cases using a simple randomization procedure (drawing identical paper lots). Group sizes varied (case 1: n=32; case 2: n=35; case 3: n=25) because of random allocation.</p></sec><sec id="s2-4"><title>Scoring Dimensions and Rubrics</title><p>Immediately after each encounter, 2 independent 100-point scores were generated using the same 50:50 weighting for information gathering and communication. Information gathering (0&#x2010;50 points) assessed completeness and relevance of the clinical history through approximately 40 checklist items. Communication (0&#x2010;50 points) assessed process quality through 10 criteria, including clinical reasoning, question appropriateness, summarization, transitional language, and humanistic care. Communication criteria were weighted from 3 to 8 points based on clinical importance; the complete rubric is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-5"><title>Statistical Analysis</title><p>Analyses were conducted in R (version 4.3.2; R Foundation for Statistical Computing). The unit of analysis was the student (n=92); each student contributed 1 paired set of AI-generated and faculty-assessed scores from a single VSP encounter. Because each encounter was scored by 1 faculty rater, clustering by rater was handled using mixed-effects models with rater-level random intercepts. Given bounded (0&#x2010;100) scores with ceiling tendencies, results were summarized using median (25th-75th percentile). All tests were 2-sided (&#x03B1;=.05).</p><p>The primary analysis used a linear mixed-effects model: human_total ~ AI_total + case + (1|rater). A random-slope specification (AI_total|rater) was explored; singular fits were interpreted as insufficient information to support between-rater slope variability. Between-rater clustering was summarized using the variance partition coefficient (VPC). Absolute agreement was summarized with intraclass correlation coefficient (ICC[2,1]; 2-way random-effects, single-measure, absolute agreement). A mixed-effects Bland-Altman model estimated mean bias and 95% limits of agreement, with tests for proportional bias. Full model specifications are in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>All analyses were repeated for information gathering and communication subscales. Prespecified sensitivity analyses refit the primary model using robust mixed-effects estimation and refit without case adjustment. Additional summaries included Spearman &#x03C1; and mean absolute error (MAE).</p></sec><sec id="s2-6"><title>Student Feedback Survey</title><p>After completing the VSP encounter, students were invited to complete an anonymous online feedback questionnaire. The survey assessed 4 dimensions: overall satisfaction (0&#x2010;10 scale), perceived convenience of the VSP platform (1&#x2010;5 Likert scale), perceived realism compared with human SP interactions (1&#x2010;5 Likert scale), and perceived feedback value relative to human SPs (1&#x2010;5 Likert scale). Responses were collected via a web-based form (Wenjuanxing) immediately after each encounter. Descriptive statistics (mean and response distribution) were computed for each item. The questionnaire was developed de novo for this study to capture platform-specific user experience and was not adapted from a previously validated instrument.</p></sec><sec id="s2-7"><title>Ethical Considerations</title><p>This study used deidentified student data from routine teaching activities. The Peking Union Medical College Hospital Ethics Committee approved this secondary analysis (I-26PJ0511) and granted a waiver of written informed consent. Students had been informed that their encounters might be used for educational research. The study adhered to the principles of the Declaration of Helsinki.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Descriptive Statistics</title><p>The study design flowchart is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. All 92 students successfully completed their VSP encounters without system crashes or encounter restarts. No encounters were excluded due to technical failures. Occasional ASR errors were informally noted by supervising faculty, primarily affecting medical terminology and rapid speech; students had been trained to correct such errors by repeating or rephrasing their questions.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flowchart of the study design. VSP: virtual standardized patient.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95578_fig01.png"/></fig><p>Ninety-two medical students each contributed 1 paired observation and were rated by 1 of 10 independent raters (8&#x2010;10 students each, median 9, IQR 9-10). Students were randomized to 3 cases (case 1: n=32; case 2: n=35; case 3: n=25). Overall scores were high, with evidence of ceiling effects: median AI total score 93.0 (IQR 89.0-95.0), median human total score 94.0 (IQR 91.0-95.0); 17 (18.5%) students scored above 95 on AI totals and 17 (18.5%) on human totals, with 6 (6.5%) students scoring above 95 on both. AI total scores ranged from 77 to 100 and human total scores from 84 to 99. Score distributions were compressed (total-score SDs 3.4-4.6: AI 4.6, human 3.4; subscale SDs 1.5-3.0: AI information 3.0, human information 2.9, AI communication 3.0, human communication 1.5). Median total scores were similar across cases, whereas rater medians varied substantially (AI 89.0&#x2010;94.0; human 87.2&#x2010;95.0), indicating meaningful between-rater severity differences (<xref ref-type="table" rid="table1">Table 1</xref>). Histograms are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Descriptive statistics of AI and human ratings across competency dimensions.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Group</td><td align="left" valign="top">AI total score</td><td align="left" valign="top">Human total score</td><td align="left" valign="top">AI information score</td><td align="left" valign="top">Human information score</td><td align="left" valign="top">AI communication score</td><td align="left" valign="top">Human communication score</td></tr></thead><tbody><tr><td align="left" valign="top">Overall (n=92), median (IQR)</td><td align="char" char="." valign="top">93.0 (89.0-95.0)</td><td align="char" char="." valign="top">94.0 (91.0-95.0)</td><td align="char" char="." valign="top">46.0 (44.0&#x2010;48.0)</td><td align="char" char="." valign="top">46.0 (44.5&#x2010;47.0)</td><td align="char" char="." valign="top">47.0 (45.0&#x2010;49.0)</td><td align="char" char="." valign="top">48.0 (47.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top" colspan="7">Case, median (IQR)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case 1 (n=32)</td><td align="char" char="." valign="top">92.0 (89.8-95.2)</td><td align="char" char="." valign="top">94.0 (91.5&#x2010;95.2)</td><td align="char" char="." valign="top">47.0 (45.0&#x2010;48.0)</td><td align="char" char="." valign="top">46.5 (44.9&#x2010;47.2)</td><td align="char" char="." valign="top">46.5 (44.8&#x2010;48.2)</td><td align="char" char="." valign="top">48.0 (47.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case 2 (n=35)</td><td align="char" char="." valign="top">93.0 (89.5&#x2010;95.0)</td><td align="char" char="." valign="top">94.0 (92.0&#x2010;95.0)</td><td align="char" char="." valign="top">46.0 (44.0&#x2010;47.0)</td><td align="char" char="." valign="top">46.0 (45.0&#x2010;46.2)</td><td align="char" char="." valign="top">47.0 (46.0-49.0)</td><td align="char" char="." valign="top">48.0 (46.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case 3 (n=25)</td><td align="char" char="." valign="top">92.0 (89.0&#x2010;94.0)</td><td align="char" char="." valign="top">93.0 (88.5&#x2010;95.0)</td><td align="char" char="." valign="top">45.0 (44.0&#x2010;47.0)</td><td align="char" char="." valign="top">45.0 (40.5&#x2010;47.0)</td><td align="char" char="." valign="top">47.0 (44.0&#x2010;48.0)</td><td align="char" char="." valign="top">48.0 (47.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top" colspan="7">Rater, median (IQR)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 1 (n=9)</td><td align="char" char="." valign="top">94.0 (91.0&#x2010;95.0)</td><td align="char" char="." valign="top">95.0 (94.0&#x2010;95.0)</td><td align="char" char="." valign="top">47.0 (45.0&#x2010;47.0)</td><td align="char" char="." valign="top">46.0 (46.0&#x2010;47.0)</td><td align="char" char="." valign="top">49.0 (46.0&#x2010;49.0)</td><td align="char" char="." valign="top">48.0 (48.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 2 (n=9)</td><td align="char" char="." valign="top">94.0 (92.0&#x2010;96.0)</td><td align="char" char="." valign="top">94.0 (93.0&#x2010;94.0)</td><td align="char" char="." valign="top">48.0 (47.0&#x2010;49.0)</td><td align="char" char="." valign="top">46.0 (44.5&#x2010;47.0)</td><td align="char" char="." valign="top">46.0 (45.0&#x2010;47.0)</td><td align="char" char="." valign="top">48.0 (47.0&#x2010;48.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 3 (n=9)</td><td align="char" char="." valign="top">93.0 (91.0&#x2010;95.0)</td><td align="char" char="." valign="top">95.0 (95.0-97.0)</td><td align="char" char="." valign="top">48.0 (46.0&#x2010;49.0)</td><td align="char" char="." valign="top">47.0 (46.0&#x2010;47.0)</td><td align="char" char="." valign="top">47.0 (44.0&#x2010;47.0)</td><td align="char" char="." valign="top">49.0 (49.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 4 (n=9)</td><td align="char" char="." valign="top">93.0 (90.0&#x2010;95.0)</td><td align="char" char="." valign="top">91.0 (90.0&#x2010;93.5)</td><td align="char" char="." valign="top">45.0 (44.0&#x2010;47.0)</td><td align="char" char="." valign="top">45.0 (44.5&#x2010;45.0)</td><td align="char" char="." valign="top">48.0 (44.0&#x2010;50.0)</td><td align="char" char="." valign="top">48.0 (46.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 5 (n=8)</td><td align="char" char="." valign="top">93.0 (89.0&#x2010;94.0)</td><td align="char" char="." valign="top">94.5 (92.6&#x2010;96.5)</td><td align="char" char="." valign="top">47.0 (45.0&#x2010;48.0)</td><td align="char" char="." valign="top">45.8 (44.0&#x2010;47.8)</td><td align="char" char="." valign="top">46.0 (43.5&#x2010;46.1)</td><td align="char" char="." valign="top">48.8 (48.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 6 (n=10)</td><td align="char" char="." valign="top">93.0 (90.5&#x2010;95.0)</td><td align="char" char="." valign="top">94.0 (93.0&#x2010;94.8)</td><td align="char" char="." valign="top">47.0 (45.2&#x2010;47.0)</td><td align="char" char="." valign="top">47.0 (47.0&#x2010;48.0)</td><td align="char" char="." valign="top">47.0 (46.2&#x2010;48.0)</td><td align="char" char="." valign="top">47.0 (46.0&#x2010;47.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 7 (n=10)</td><td align="char" char="." valign="top">91.0 (89.2&#x2010;93.5)</td><td align="char" char="." valign="top">87.2 (86.2&#x2010;91.2)</td><td align="char" char="." valign="top">44.0 (41.2&#x2010;45.0)</td><td align="char" char="." valign="top">40.5 (39.0&#x2010;44.5)</td><td align="char" char="." valign="top">47.5 (45.5&#x2010;49.8)</td><td align="char" char="." valign="top">47.0 (46.0&#x2010;47.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 8 (n=10)</td><td align="char" char="." valign="top">92.5 (88.2&#x2010;94.8)</td><td align="char" char="." valign="top">91.2 (90.2&#x2010;92.0)</td><td align="char" char="." valign="top">45.0 (43.2&#x2010;45.8)</td><td align="char" char="." valign="top">44.8 (44.0&#x2010;45.8)</td><td align="char" char="." valign="top">48.5 (47.2&#x2010;49.0)</td><td align="char" char="." valign="top">46.5 (46.0&#x2010;47.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 9 (n=8)</td><td align="char" char="." valign="top">93.0 (92.0&#x2010;95.2)</td><td align="char" char="." valign="top">95.0 (93.5&#x2010;96.2)</td><td align="char" char="." valign="top">47.0 (46.0&#x2010;47.0)</td><td align="char" char="." valign="top">47.0 (46.8&#x2010;48.0)</td><td align="char" char="." valign="top">47.0 (45.8&#x2010;48.2)</td><td align="char" char="." valign="top">48.0 (46.0&#x2010;49.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rater 10 (n=10)</td><td align="char" char="." valign="top">89.0 (86.5&#x2010;91.2)</td><td align="char" char="." valign="top">92.0 (89.2&#x2010;95.8)</td><td align="char" char="." valign="top">44.0 (42.2&#x2010;45.0)</td><td align="char" char="." valign="top">44.0 (40.2&#x2010;45.8)</td><td align="char" char="." valign="top">44.5 (43.2&#x2010;47.8)</td><td align="char" char="." valign="top">49.0 (49.0&#x2010;49.8)</td></tr></tbody></table></table-wrap></sec><sec id="s3-2"><title>Rater Clustering</title><p>A linear mixed-effects model with fixed effects for the AI total score and case and a random intercept for rater showed notable rater dependence (Var[rater]=3.05; Var[residual]=5.29), yielding a VPC of 0.37. This indicates that 37% of the unexplained variance in human total scores was attributable to between-rater differences in average scoring severity or leniency. Accounting for rater clustering improved fit versus ordinary least squares (&#x0394;Akaike information criterion=19.10), supporting the presence of nontrivial rater heterogeneity [<xref ref-type="bibr" rid="ref11">11</xref>].</p></sec><sec id="s3-3"><title>Agreement Between AI and Human Total Scores</title><p>In the primary mixed model, the AI total score was positively associated with the human total score (&#x03B2;=0.37, 95% CI 0.26&#x2010;0.48; <italic>P</italic>&#x003C;.001; <xref ref-type="table" rid="table2">Table 2</xref> and <xref ref-type="fig" rid="figure2">Figure 2</xref>). Case indicators were not statistically significant (Case 2: <italic>P</italic>=.86; Case 3: <italic>P</italic>=.23). Agreement was moderate: Spearman &#x03C1;=0.50, ICC(2,1)=0.51, and MAE=3.11 points. Mixed-effects Bland-Altman analysis showed a mean bias that did not reach statistical significance (1.26 points, 95% CI &#x2212;0.48 to 3.01; <italic>P</italic>=.16) but wide 95% limits of agreement (&#x2212;4.95 to 7.48; width=12.43 points), indicating educationally meaningful individual-level discrepancies. Proportional bias was present (&#x03B2;_proportional bias=&#x2212;0.55; <italic>P</italic>&#x003C;.001), with AI assigning relatively higher scores at the lower end of the performance distribution (<xref ref-type="fig" rid="figure3">Figure 3</xref>).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Model-based association and AI-human agreement.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">LMM<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>, &#x03B2; (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Spearman &#x03C1;</td><td align="left" valign="bottom">ICC<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>(2,1)</td><td align="left" valign="bottom">Mean bias (human&#x2013;AI)</td><td align="left" valign="bottom">Conditional 95% LoA<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="bottom">Proportional bias, &#x03B2;_proportional bias</td><td align="left" valign="bottom">Proportional bias, <italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Total score</td><td align="char" char="." valign="top">0.37 (0.26&#x2010;0.48)</td><td align="char" char="." valign="top">&#x003C;.001</td><td align="char" char="." valign="top">0.50</td><td align="char" char="." valign="top">0.51</td><td align="char" char="." valign="top">1.26</td><td align="char" char="." valign="top">&#x2212;4.95 to 7.48</td><td align="char" char="." valign="top">&#x2212;0.55</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Information</td><td align="char" char="." valign="top">0.46 (0.30&#x2010;0.63)</td><td align="char" char="." valign="top">&#x003C;.001</td><td align="char" char="." valign="top">0.49</td><td align="char" char="." valign="top">0.54</td><td align="char" char="." valign="top">&#x2212;0.38</td><td align="char" char="." valign="top">&#x2212;5.52 to 4.77</td><td align="char" char="." valign="top">&#x2212;0.10</td><td align="char" char="." valign="top">.40</td></tr><tr><td align="left" valign="top">Communication</td><td align="char" char="." valign="top">0.27 (0.19&#x2010;0.34)</td><td align="char" char="." valign="top">&#x003C;.001</td><td align="char" char="." valign="top">0.28</td><td align="char" char="." valign="top">0.29</td><td align="char" char="." valign="top">1.29</td><td align="char" char="." valign="top">&#x2212;1.69 to 4.28</td><td align="char" char="." valign="top">&#x2212;0.94</td><td align="char" char="." valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>LMM: linear mixed-effects model.</p></fn><fn id="table2fn2"><p><sup>b</sup>ICC: intraclass correlation coefficient. </p></fn><fn id="table2fn3"><p><sup>c</sup>LoA: limits of agreement.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Scatter plot showing the association between AI-generated total scores and human-assessed total scores. Each point represents an individual medical student&#x2019;s score pair (n=92). The solid black line indicates the linear regression fit from the mixed-effects model (&#x03B2;=0.37, 95% CI 0.26&#x2010;0.48; <italic>P</italic>&#x003C;.001), and the gray shaded area represents the 95% CI for the regression line.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95578_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Mixed-effects Bland-Altman plot for AI and human total scores. The solid red line indicates the mean bias (1.26 points, 95% CI &#x2212;0.48 to 3.01), which did not reach statistical significance at the group level (<italic>P</italic>=.16). The dashed red lines represent the 95% limits of agreement (LoA; &#x2212;4.95 to 7.48; width=12.43 points). The blue line shows proportional bias (&#x03B2;_proportional bias=&#x2212;0.55, 95% CI &#x2212;0.75 to &#x2212;0.35; <italic>P</italic>&#x003C;.001), indicating that AI assigned relatively higher scores at the lower end of the performance distribution.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95578_fig03.png"/></fig></sec><sec id="s3-4"><title>Subgroup Patterns</title><p>Rater-stratified agreement varied widely (ICC[2,1] estimates 0.57&#x2010;0.93; Spearman &#x03C1; 0.21&#x2010;0.93; MAE 2.17&#x2010;5.00) with imprecise estimates (n=8&#x2010;10 per rater). Mean AI-human bias also varied in direction across raters (&#x2013;2.05 to 3.25), consistent with rater-specific scoring stringency or leniency or local calibration effects (<xref ref-type="fig" rid="figure4">Figure 4</xref>).</p><p>Case-stratified results suggested highest agreement in case 1 (ICC=0.63; &#x03C1;=0.62; MAE=2.83) and lowest in case 3 (ICC=0.36; &#x03C1;=0.32; MAE=4.24). These case-level estimates should be interpreted descriptively given limited per-subgroup sample sizes.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Forest (dot-whisker) plot of rater-level heterogeneity in AI-human agreement for total scores. Each row represents an independent rater (n=8&#x2010;10), ordered by ICC(2,1) point estimate. Points show estimates and whiskers show 95% CIs for ICC(2,1), bias (mean Human&#x2013;AI; vertical line at 0), and mean absolute error (MAE). Given small per-rater sample sizes, results are descriptive. ICC: intraclass correlation coefficient.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95578_fig04.png"/></fig></sec><sec id="s3-5"><title>Sensitivity Analyses</title><p>Prespecified sensitivity analyses examined model robustness. Refitting the primary mixed-effects model with robust Huber estimation yielded a nearly identical AI coefficient (&#x03B2;=0.35, 95% CI 0.24&#x2010;0.46; <italic>P</italic>&#x003C;.001). Omitting case adjustment also produced an essentially unchanged estimate (&#x03B2;=0.36, 95% CI 0.25&#x2010;0.47; <italic>P</italic>&#x003C;.001). Full details are provided in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p></sec><sec id="s3-6"><title>Subscale Analyses</title><p>AI-human agreement differed by subscale, with stronger alignment for information gathering than for communication (<xref ref-type="table" rid="table2">Table 2</xref>). For information gathering, &#x03B2;=0.46 (<italic>P</italic>&#x003C;.001); &#x03C1;=0.49; ICC=0.54 (95% CI 0.38&#x2010;0.67); VPC=0.23. For communication, &#x03B2;=0.27 (<italic>P</italic>&#x003C;.001); &#x03C1;=0.28; ICC=0.29 (95% CI 0.08&#x2010;0.47); VPC=0.52, indicating greater rater dependence and lower reproducibility.</p></sec><sec id="s3-7"><title>Student Feedback</title><p>Of the 92 participating students, 63 (68.5%) completed the postencounter feedback questionnaire. Overall satisfaction was moderate (mean 6.4, SD 1.9 on a 10-point scale). Students highly valued the convenience of the VSP platform (mean 4.1, SD 0.8 on a 5-point scale). However, perceived realism was low (mean 1.8, SD 0.9 on a 5-point scale), and perceived feedback value relative to human SPs was below the scale midpoint (mean 2.4, SD 1.1 on a 5-point scale). Full survey results, including item-level response distributions, are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study of a voice-based, LLM-powered VSP used by fourth-year medical students, 3 main findings emerged. First, AI-generated and faculty-assigned total scores showed moderate agreement, with stronger alignment for information gathering than for communication. Second, rater differences accounted for 37% of residual variance in total scores and 52% in communication scores, indicating that apparent AI-human disagreement partly reflects human-human variability. Third, proportional bias indicated that AI tended to assign relatively higher scores at the lower end of the performance distribution.</p></sec><sec id="s4-2"><title>Interpretation and Psychometric Implications</title><p>From a psychometric perspective, the magnitude of AI-human agreement observed here does not support unsupervised, high-stakes use for individual learners. Although mean bias was small and did not reach statistical significance (1.26 points, 95% CI &#x2212;0.48 to 3.01; <italic>P</italic>=.16), and the 95% limits of agreement spanned 12.43 points on a 100-point scale, wide enough that an individual learner could plausibly be classified differently depending on whether the AI or a faculty rater provided the score. Ceiling effects further complicate interpretation: with 18% of students scoring above 95 and total score SDs of only 3.4 to 4.6 points, the observed ICC of 0.51 should be treated as a lower-bound estimate. A range-restriction sensitivity analysis (Thorndike Case 2 correction) yielded a corrected ICC of approximately 0.72 (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Proportional bias (AI assigning relatively higher scores to lower-performing students) further cautions against high-stakes use, particularly for communication skills.</p><p>At the same time, the overall direction and robustness of the AI-human association, which were stronger than those often observed between individual human raters in clinical performance assessments, suggested practical utility for low-stakes formative contexts [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. In these contexts, the primary goal is not precise interchangeability with a single rater but scalable feedback and increased sampling of performance.</p></sec><sec id="s4-3"><title>Validity Argument</title><p>This study provided criterion-related validity evidence by quantifying associations with faculty judgments. However, it addressed psychometric feasibility (score agreement) rather than educational feasibility (impact on learning outcomes), which requires dedicated longitudinal studies. Validity also depends on content alignment, response process coherence, and consequences [<xref ref-type="bibr" rid="ref6">6</xref>]; these domains require further investigation before broader adoption&#x2014;especially for higher-stakes purposes&#x2014;can be justified.</p></sec><sec id="s4-4"><title>Subscale-Specific Performance and Equity</title><p>Information gathering showed higher AI-human agreement and lower rater variance than communication. This aligned with the relative objectivity of checklist-based content compared with the judgment-laden nature of communication behaviors [<xref ref-type="bibr" rid="ref14">14</xref>]. In practice, this makes AI scoring potentially useful for identifying omitted history elements, quantifying completeness, and tracking progress over time.</p><p>Communication scoring, however, remains challenging. Subjective judgments about rapport, empathy, and professionalism depend on nuanced, context-sensitive interpretation that current LLMs may not reliably replicate [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. This is also an equity-relevant domain: communication judgments&#x2014;whether by humans or AI&#x2014;may be sensitive to accent, language proficiency, culturally patterned interaction styles, or disability-related differences. These potential differential effects should be tested explicitly rather than assumed away [<xref ref-type="bibr" rid="ref17">17</xref>].</p></sec><sec id="s4-5"><title>Rater Heterogeneity and Faculty Development</title><p>Rater-stratified results showed wide variability in AI-human agreement across faculty, reinforcing that human raters themselves are not interchangeable. Because students were nested within raters in this design, a single human score cannot be treated as an error-free reference; rather, it reflects one draw from a distribution of possible faculty judgments. This variability also limited generalizability: AI-human agreement observed here depended on the particular mix of raters and their scoring tendencies [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>Rather than treating faculty ratings as a fixed &#x201C;gold standard,&#x201D; a more defensible target is whether AI scores fall within the range of acceptable human variation. If AI-human disagreement is comparable to human-human disagreement, AI scores can be considered psychometrically equivalent to an additional human rater&#x2014;a standard that has been proposed for automated scoring in other domains [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>].</p></sec><sec id="s4-6"><title>Relation to Prior Work</title><p>Prior work on LLMs in medical education has emphasized written examination performance [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>]. This study instead evaluated an LLM as an assessor of interactive clinical performance, foregrounding questions of rater variability and validity for performance scores. Our results aligned with emerging evidence that AI-human agreement in clinical communication assessment is moderate rather than excellent [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref26">26</xref>], and they extended that literature by quantifying how much apparent AI-human disagreement reflected human-human variability [<xref ref-type="bibr" rid="ref6">6</xref>]. Importantly, moderate AI-human agreement does not necessarily imply invalid AI scoring; rather, it highlights that both AI and human raters bring distinct sources of error and bias [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>].</p></sec><sec id="s4-7"><title>Limitations</title><p>This study has several limitations. Data were drawn from a single institution with predominantly high-performing students, and ceiling effects may limit generalizability to other learner populations. Each student completed only 1 encounter evaluated by a single faculty rater, precluding assessment of within-student stability and human interrater reliability. Faculty received brief rubric orientation without structured calibration, which may have introduced variability in scoring standards. Although we examined quantitative score agreement, we did not evaluate the quality or educational impact of AI-generated narrative feedback. Students&#x2019; awareness of interacting with an AI system may have influenced communication behaviors. ASR error rates were not systematically recorded, limiting assessment of potential transcription bias. The scoring prompt was aligned with local faculty norms, and the analyses relied on specific LLM models, prompts, and cases; these factors may limit generalizability across institutions and over time, underscoring the need for multisite validation. Finally, the student feedback questionnaire was newly developed for this study and has not undergone formal psychometric validation (eg, content validity or reliability testing); the survey results should therefore be interpreted as exploratory and descriptive.</p></sec><sec id="s4-8"><title>Student Experience and Perceived Utility</title><p>Student feedback on the VSP platform highlighted a tension between convenience and authenticity. While learners appreciated the on-demand accessibility of AI-based practice, they rated the realism of AI-simulated patient interactions and the perceived value of AI-generated feedback lower than comparable ratings for human SP encounters. These perceptions likely reflect current limitations in conversational naturalness, which may attenuate engagement and learning transfer. Improving dialogue fidelity and closing the perceived feedback quality gap represent key priorities for VSP refinement. The juxtaposition of moderate satisfaction with low perceived realism suggests that learner experience is shaped by factors beyond scoring accuracy alone, underscoring the need for multidimensional evaluation of VSP systems.</p></sec><sec id="s4-9"><title>Practical Implications</title><p>Given the current evidence, we suggest that LLM-based scoring in VSPs can be used to prioritize practice opportunities and deliver rapid feedback&#x2014;especially for information gathering&#x2014;where scalability and frequent practice are central and where perfect interchangeability with a faculty score is not required. High-stakes use should remain human-supervised and, if used at all, should be embedded in hybrid workflows where AI provides an additional perspective rather than a final decision.</p><p>Within programmatic assessment, AI scores may be most defensible for increasing sampling density, flagging learners for coaching, and triangulating with other evidence (OSCEs, workplace-based assessments, and supervisor narratives), rather than serving as a standalone progression metric.</p></sec><sec id="s4-10"><title>Future Directions</title><p>Future work should develop behavioral benchmarks that disentangle AI error from legitimate variation in faculty judgment, characterize differential item functioning across student subgroups, and test rater calibration interventions that leverage AI scores to improve human rating consistency.</p></sec><sec id="s4-11"><title>Conclusions</title><p>In this undergraduate medical course, an LLM-based VSP generated history-taking and communication scores that showed moderate agreement with human raters, with greater convergence for information gathering than for communication skills. Substantial variability among human raters indicated that human scoring was itself an imperfect reference standard. Overall, these findings supported LLM-based scoring as a feasible, scalable tool for formative assessment and practice, while cautioning against unsupervised high-stakes application. Ongoing validation should prioritize rater calibration, communication scoring refinement, and equity auditing as these technologies mature in medical education.</p></sec></sec></body><back><ack><p>The authors thank the faculty raters and students who participated in this study.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the Undergraduate Education and Teaching Reform Project of Peking Union Medical College (grant 2025bkjg011).</p></sec><sec><title>Data Availability</title><p>The deidentified dataset supporting the findings of this study is openly available on Zenodo at [<xref ref-type="bibr" rid="ref27">27</xref>]. The dataset includes student-level AI and faculty scores, case assignments, rater identifiers, and item-level scoring details.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: XG, XH, XS</p><p>Data curation: XG</p><p>Formal analysis: XG</p><p>Funding acquisition: XS</p><p>Investigation: XG, RH, LZ, HL, BZ, CW, WQ, MZ</p><p>Methodology: XS</p><p>Project administration: XG</p><p>Resources: XS</p><p>Software: XS</p><p>Supervision: XH, XS</p><p>Validation: XS</p><p>Visualization: XS</p><p>Writing&#x2014;original draft: XG</p><p>Writing&#x2014;review and editing: XG, XH, XS</p><p>All authors approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ASR</term><def><p>automatic speech recognition</p></def></def-item><def-item><term id="abb2">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb4">MAE</term><def><p>mean absolute error</p></def></def-item><def-item><term id="abb5">OSCE</term><def><p>objective structured clinical examination</p></def></def-item><def-item><term id="abb6">SP</term><def><p>standardized patient</p></def></def-item><def-item><term id="abb7">VPC</term><def><p>variance partition coefficient</p></def></def-item><def-item><term id="abb8">VSP</term><def><p>virtual standardized patient</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kononowicz</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Zary</surname><given-names>N</given-names> </name><name name-style="western"><surname>Edelbring</surname><given-names>S</given-names> </name><name name-style="western"><surname>Corral</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hege</surname><given-names>I</given-names> </name></person-group><article-title>Virtual patients--what are we talking about? A framework to classify the meanings of the term in healthcare education</article-title><source>BMC Med Educ</source><year>2015</year><month>02</month><day>1</day><volume>15</volume><fpage>11</fpage><pub-id pub-id-type="doi">10.1186/s12909-015-0296-3</pub-id><pub-id pub-id-type="medline">25638167</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kononowicz</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Woodham</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Edelbring</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Virtual patient simulations in health professions education: systematic review and meta-analysis by the digital health education collaboration</article-title><source>J Med Internet Res</source><year>2019</year><month>07</month><day>2</day><volume>21</volume><issue>7</issue><fpage>e14676</fpage><pub-id pub-id-type="doi">10.2196/14676</pub-id><pub-id pub-id-type="medline">31267981</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cook</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Triola</surname><given-names>MM</given-names> </name></person-group><article-title>Virtual patients: a critical literature review and proposed next steps</article-title><source>Med Educ</source><year>2009</year><month>04</month><volume>43</volume><issue>4</issue><fpage>303</fpage><lpage>311</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2923.2008.03286.x</pub-id><pub-id pub-id-type="medline">19335571</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abd-Alrazaq</surname><given-names>A</given-names> </name><name name-style="western"><surname>AlSaad</surname><given-names>R</given-names> </name><name name-style="western"><surname>Alhuwail</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Large language models in medical education: opportunities, challenges, and future directions</article-title><source>JMIR Med Educ</source><year>2023</year><month>06</month><day>1</day><volume>9</volume><fpage>e48291</fpage><pub-id pub-id-type="doi">10.2196/48291</pub-id><pub-id pub-id-type="medline">37261894</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jukiewicz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wyrwa</surname><given-names>M</given-names> </name></person-group><article-title>Can ChatGPT replace the teacher in assessment? A review of research on the use of large language models in grading and providing feedback</article-title><source>Appl Sci</source><year>2026</year><volume>16</volume><issue>2</issue><fpage>680</fpage><pub-id pub-id-type="doi">10.3390/app16020680</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cook</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Zendejas</surname><given-names>B</given-names> </name><name name-style="western"><surname>Hamstra</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Hatala</surname><given-names>R</given-names> </name><name name-style="western"><surname>Brydges</surname><given-names>R</given-names> </name></person-group><article-title>What counts as validity evidence? Examples and prevalence in a systematic review of simulation-based assessment</article-title><source>Adv Health Sci Educ Theory Pract</source><year>2014</year><month>05</month><volume>19</volume><issue>2</issue><fpage>233</fpage><lpage>250</lpage><pub-id pub-id-type="doi">10.1007/s10459-013-9458-4</pub-id><pub-id pub-id-type="medline">23636643</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Homer</surname><given-names>M</given-names> </name></person-group><article-title>Pass/fail decisions and standards: the impact of differential examiner stringency on OSCE outcomes</article-title><source>Adv Health Sci Educ Theory Pract</source><year>2022</year><month>05</month><volume>27</volume><issue>2</issue><fpage>457</fpage><lpage>473</lpage><pub-id pub-id-type="doi">10.1007/s10459-022-10096-9</pub-id><pub-id pub-id-type="medline">35230590</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brannick</surname><given-names>MT</given-names> </name><name name-style="western"><surname>Erol-Korkmaz</surname><given-names>HT</given-names> </name><name name-style="western"><surname>Prewett</surname><given-names>M</given-names> </name></person-group><article-title>A systematic review of the reliability of objective structured clinical examination scores</article-title><source>Med Educ</source><year>2011</year><month>12</month><volume>45</volume><issue>12</issue><fpage>1181</fpage><lpage>1189</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2923.2011.04075.x</pub-id><pub-id pub-id-type="medline">21988659</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tekin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yurdal</surname><given-names>MO</given-names> </name><name name-style="western"><surname>Toraman</surname><given-names>&#x00C7;</given-names> </name><name name-style="western"><surname>Korkmaz</surname><given-names>G</given-names> </name><name name-style="western"><surname>Uysal</surname><given-names>&#x0130;</given-names> </name></person-group><article-title>Is AI the future of evaluation in medical education?? AI vs. human evaluation in objective structured clinical examination</article-title><source>BMC Med Educ</source><year>2025</year><month>05</month><day>1</day><volume>25</volume><issue>1</issue><fpage>641</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-07241-4</pub-id><pub-id pub-id-type="medline">40312328</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Schi&#x00F6;tt</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ivegren</surname><given-names>W</given-names> </name><name name-style="western"><surname>Borg</surname><given-names>A</given-names> </name><name name-style="western"><surname>Parodis</surname><given-names>I</given-names> </name><name name-style="western"><surname>Skantze</surname><given-names>G</given-names> </name></person-group><article-title>Using LLMs to grade clinical reasoning for medical students in virtual patient dialogues</article-title><source>Proceedings of the 26th Annual Meeting of the Special Interest Group on Discourse and Dialogue</source><year>2025</year><access-date>2026-08-12</access-date><publisher-name>Association for Computational Linguistics</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.sigdial-1.56/">https://aclanthology.org/2025.sigdial-1.56/</ext-link></comment></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burnham</surname><given-names>KP</given-names> </name><name name-style="western"><surname>Anderson</surname><given-names>DR</given-names> </name></person-group><article-title>Multimodel inference: understanding AIC and BIC in model selection</article-title><source>Sociol Methods Res</source><year>2004</year><volume>33</volume><issue>2</issue><fpage>261</fpage><lpage>304</lpage><pub-id pub-id-type="doi">10.1177/0049124104268644</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yamamoto</surname><given-names>A</given-names> </name><name name-style="western"><surname>Koda</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ogawa</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Enhancing medical interview skills through AI-simulated patient interactions: nonrandomized controlled trial</article-title><source>JMIR Med Educ</source><year>2024</year><month>09</month><day>23</day><volume>10</volume><fpage>e58753</fpage><pub-id pub-id-type="doi">10.2196/58753</pub-id><pub-id pub-id-type="medline">39312284</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Development and validation of a large language model-based system for medical history-taking training: prospective multicase study on evaluation stability, human-AI consistency, and transparency</article-title><source>JMIR Med Educ</source><year>2025</year><month>08</month><day>29</day><volume>11</volume><fpage>e73419</fpage><pub-id pub-id-type="doi">10.2196/73419</pub-id><pub-id pub-id-type="medline">40882613</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>K</given-names> </name><name name-style="western"><surname>Giorgi</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Mani</surname><given-names>P</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>C</given-names> </name></person-group><article-title>From feedback to checklists: grounded evaluation of AI-generated clinical notes</article-title><source>Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing</source><year>2025</year><publisher-name>Association for Computational Linguistics</publisher-name><pub-id pub-id-type="doi">10.18653/v1/2025.emnlp-industry.104</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dorrestein</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ritter</surname><given-names>C</given-names> </name><name name-style="western"><surname>De Mol</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Validity evidence for communication skills assessment in health professions education: a scoping review</article-title><source>BMJ Open</source><year>2025</year><month>09</month><day>5</day><volume>15</volume><issue>9</issue><fpage>e096799</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2024-096799</pub-id><pub-id pub-id-type="medline">40912699</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hodges</surname><given-names>B</given-names> </name></person-group><article-title>Assessment in the post-psychometric era: learning to love the subjective and collective</article-title><source>Med Teach</source><year>2013</year><month>07</month><volume>35</volume><issue>7</issue><fpage>564</fpage><lpage>568</lpage><pub-id pub-id-type="doi">10.3109/0142159X.2013.789134</pub-id><pub-id pub-id-type="medline">23631408</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cleland</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Knight</surname><given-names>LV</given-names> </name><name name-style="western"><surname>Rees</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Tracey</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bond</surname><given-names>CM</given-names> </name></person-group><article-title>Is it me or is it them? Factors that influence the passing of underperforming students</article-title><source>Med Educ</source><year>2008</year><month>08</month><volume>42</volume><issue>8</issue><fpage>800</fpage><lpage>809</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2923.2008.03113.x</pub-id><pub-id pub-id-type="medline">18715477</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Anthony</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Styck</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Volpe</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Robert</surname><given-names>CR</given-names> </name></person-group><article-title>Using many-facet Rasch measurement and generalizability theory to explore rater effects for direct behavior rating-multi-item scales</article-title><source>Sch Psychol</source><year>2023</year><month>03</month><volume>38</volume><issue>2</issue><fpage>119</fpage><lpage>128</lpage><pub-id pub-id-type="doi">10.1037/spq0000518</pub-id><pub-id pub-id-type="medline">36174169</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Uto</surname><given-names>M</given-names> </name></person-group><article-title>A Bayesian many-facet Rasch model with Markov modeling for rater severity drift</article-title><source>Behav Res Methods</source><year>2023</year><month>10</month><volume>55</volume><issue>7</issue><fpage>3910</fpage><lpage>3928</lpage><pub-id pub-id-type="doi">10.3758/s13428-022-01997-z</pub-id><pub-id pub-id-type="medline">36284065</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Roberts</surname><given-names>RH</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Hutchings</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Dobbs</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Whitaker</surname><given-names>IS</given-names> </name></person-group><article-title>Comparative study of ChatGPT and human evaluators on the assessment of medical literature according to recognised reporting standards</article-title><source>BMJ Health Care Inform</source><year>2023</year><month>10</month><volume>30</volume><issue>1</issue><fpage>e100830</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2023-100830</pub-id><pub-id pub-id-type="medline">37827724</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Prentice</surname><given-names>S</given-names> </name><name name-style="western"><surname>Benson</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kirkpatrick</surname><given-names>E</given-names> </name><name name-style="western"><surname>Schuwirth</surname><given-names>L</given-names> </name></person-group><article-title>Workplace-based assessments in postgraduate medical education: a hermeneutic review</article-title><source>Med Educ</source><year>2020</year><month>11</month><volume>54</volume><issue>11</issue><fpage>981</fpage><lpage>992</lpage><pub-id pub-id-type="doi">10.1111/medu.14221</pub-id><pub-id pub-id-type="medline">32403200</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hyde</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fessey</surname><given-names>C</given-names> </name><name name-style="western"><surname>Boursicot</surname><given-names>K</given-names> </name><name name-style="western"><surname>MacKenzie</surname><given-names>R</given-names> </name><name name-style="western"><surname>McGrath</surname><given-names>D</given-names> </name></person-group><article-title>OSCE rater cognition - an international multi-centre qualitative study</article-title><source>BMC Med Educ</source><year>2022</year><month>01</month><day>3</day><volume>22</volume><issue>1</issue><fpage>6</fpage><pub-id pub-id-type="doi">10.1186/s12909-021-03077-w</pub-id><pub-id pub-id-type="medline">34980099</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kung</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Cheatham</surname><given-names>M</given-names> </name><name name-style="western"><surname>Medenilla</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title><source>PLOS Digit Health</source><year>2023</year><month>02</month><volume>2</volume><issue>2</issue><fpage>e0000198</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id><pub-id pub-id-type="medline">36812645</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gilson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Safranek</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>How does ChatGPT perform on the United States Medical Licensing Examination (USMLE)? The implications of large language models for medical education and knowledge assessment</article-title><source>JMIR Med Educ</source><year>2023</year><month>02</month><day>8</day><volume>9</volume><fpage>e45312</fpage><pub-id pub-id-type="doi">10.2196/45312</pub-id><pub-id pub-id-type="medline">36753318</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burke</surname><given-names>HB</given-names> </name><name name-style="western"><surname>Hoang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lopreiato</surname><given-names>JO</given-names> </name><etal/></person-group><article-title>Assessing the ability of a large language model to score free-text medical student clinical notes: quantitative study</article-title><source>JMIR Med Educ</source><year>2024</year><month>07</month><day>25</day><volume>10</volume><fpage>e56342</fpage><pub-id pub-id-type="doi">10.2196/56342</pub-id><pub-id pub-id-type="medline">39118469</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dhillon</surname><given-names>IK</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>GK</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>S</given-names> </name></person-group><article-title>Evaluating the pediatric behavior guidance of students based on actual clinical transcripts scored by faculty and large language models: pilot comparative study</article-title><source>JMIR Med Educ</source><year>2026</year><month>06</month><day>12</day><volume>12</volume><fpage>e83376</fpage><pub-id pub-id-type="doi">10.2196/83376</pub-id><pub-id pub-id-type="medline">42285034</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>X</given-names> </name></person-group><article-title>AI score and human score of medical students&#x2019; VSP interview</article-title><source>Zenodo</source><year>2026</year><access-date>2026-08-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/20774435">https://zenodo.org/records/20774435</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Complete large language model scoring prompt and iterative calibration procedure.</p><media xlink:href="mededu_v12i1e95578_app1.docx" xlink:title="DOCX File, 30 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Detailed statistical model specifications.</p><media xlink:href="mededu_v12i1e95578_app2.docx" xlink:title="DOCX File, 39 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Score distribution histograms and student feedback survey results.</p><media xlink:href="mededu_v12i1e95578_app3.docx" xlink:title="DOCX File, 389 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Rater-stratified agreement metrics, sensitivity analyses, and lower-quartile analysis.</p><media xlink:href="mededu_v12i1e95578_app4.docx" xlink:title="DOCX File, 32 KB"/></supplementary-material></app-group></back></article>