<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id><journal-id journal-id-type="publisher-id">mededu</journal-id><journal-id journal-id-type="index">20</journal-id><journal-title>JMIR Medical Education</journal-title><abbrev-journal-title>JMIR Med Educ</abbrev-journal-title><issn pub-type="epub">2369-3762</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v12i1e96819</article-id><article-id pub-id-type="doi">10.2196/96819</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>What Platform Scores Miss: Multidimensional Evaluation of AI Teaching Agents in Medical Education</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Hui</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Qu</surname><given-names>Lihui</given-names></name><degrees>PHD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zheng</surname><given-names>Jianmin</given-names></name><degrees>MMgmt</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xiong</surname><given-names>Yi</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Bai</surname><given-names>Hongbo</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ji</surname><given-names>Ruiying</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Guohui</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Wanling</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cheng</surname><given-names>Zirui</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Youbang</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Yang</surname><given-names>Chun-tao</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>Office of Academic Affairs, Guangzhou Medical University</institution><addr-line>Guangzhou</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Physiology, School of Basic Medical Sciences, The Fourth Affiliated Hospital of Guangzhou Medical University</institution><addr-line>Guangzhou</addr-line><country>China</country></aff><aff id="aff3"><institution>Guangzhou Municipal and Guangdong Provincial Key Laboratory of Protein Modification and Disease, Department of Physiology, School of Basic Medical Sciences, Guangzhou Medical University</institution><addr-line>1 Xinzao Road</addr-line><addr-line>Guangzhou</addr-line><addr-line>Guangdong</addr-line><country>China</country></aff><aff id="aff4"><institution>Department of Endocrinology, The First Affiliated Hospital of Guangzhou Medical University</institution><addr-line>Guangzhou</addr-line><country>China</country></aff><aff id="aff5"><institution>Zhongshan School of Medicine, Sun Yat-Sen University</institution><addr-line>Guangzhou</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Jerjes</surname><given-names>Waseem</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Liu</surname><given-names>Song</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yang</surname><given-names>Yunchu</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Chun-tao Yang, PhD, Guangzhou Municipal and Guangdong Provincial Key Laboratory of Protein Modification and Disease, Department of Physiology, School of Basic Medical Sciences, Guangzhou Medical University, 1 Xinzao Road, Guangzhou, Guangdong, 511436, China, 86 20-37103213; <email>cyang@gzhmu.edu.cn</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>3</day><month>9</month><year>2026</year></pub-date><volume>12</volume><elocation-id>e96819</elocation-id><history><date date-type="received"><day>01</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>10</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>16</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Hui Zhang, Lihui Qu, Jianmin Zheng, Yi Xiong, Hongbo Bai, Ruiying Ji, Guohui Liu, Wanling Chen, Zirui Cheng, Youbang Chen, Chun-tao Yang. Originally published in JMIR Medical Education (<ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org">https://mededu.jmir.org</ext-link>), 3.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org/">https://mededu.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mededu.jmir.org/2026/1/e96819"/><abstract><sec><title>Background</title><p>Large language model (LLM)&#x2013;based AI teaching agents are increasingly used in medical education, yet their pedagogical quality is typically judged by platform-generated scores whose scoring criteria are undisclosed and may not reflect the teaching quality of the agent.</p></sec><sec><title>Objective</title><p>This study aimed to develop and validate a multidimensional rubric for evaluating AI teaching agents and to examine the correspondence between platform scores and rubric-based teaching quality.</p></sec><sec sec-type="methods"><title>Methods</title><p>Eight AI teaching agents covering an endocrinology curriculum were deployed across 4 role-play paradigms (patient, student, expert, and family). Twenty-two fourth-year medical students generated 167 dialogues, which were scored both by the platform and by an independently applied 8-dimension rubric (100 points, covering knowledge accuracy, pedagogical guidance, knowledge coverage, role-play quality, difficulty calibration, medical safety, student engagement, and feedback quality). Each dialogue was scored 4 times by a primary evaluator (Claude Opus 4.8; mean within-model SD 0.36), with 2 additional LLMs as robustness checks; 40 dialogues spanning all agents were rescored by a medical-education expert for validation.</p></sec><sec sec-type="results"><title>Results</title><p>Platform and rubric rankings diverged for most agents: the agent ranked third by the platform ranked last on rubric-based quality, and the platform&#x2019;s fourth-ranked agent ranked first. Agents differed most on knowledge-related dimensions (knowledge coverage coefficient of variation=27.3%) and least on role-play quality (coefficient of variation=5.7%), while difficulty calibration was a shared weakness. In a case-level observation, one agent revised specifically to strengthen empathy attained high role-play quality yet the lowest knowledge coverage of all agents. AI scores agreed with expert ratings at the total-score level (intraclass correlation coefficient=0.51) and on cognitive-process dimensions, but agreement was low for the more subjective dimensions. Student gender showed no detectable effect, though this analysis was underpowered.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this exploratory study, platform-generated scores reflected a construct different from agent teaching quality and should be used as a complement rather than as the sole quality indicator. The 8-dimension rubric provides a transparent, standardized alternative that reveals differences missed by platform scores, including a lack of association between empathy and knowledge coverage that warrants attention in future agent design.</p></sec></abstract><kwd-group><kwd>AI teaching agents</kwd><kwd>educational data science</kwd><kwd>educational evaluation</kwd><kwd>endocrinology</kwd><kwd>large language models</kwd><kwd>LLMs</kwd><kwd>multidimensional rubric</kwd><kwd>precision education</kwd><kwd>role-play</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Bridging the gap between theoretical knowledge and clinical competence remains a central challenge in medical education, particularly in disciplines requiring multisystem integration such as endocrinology [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Simulated patient encounters are among the most effective approaches for developing clinical reasoning and communication skills, but their scalability is constrained by the cost and limited availability of standardized human patients [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. Large language model (LLM)&#x2013;based AI agents have emerged as a promising alternative, achieving clinical fidelity comparable to standardized human patients at substantially lower cost [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref13">13</xref>], and they are now being adopted in medical curricula at a pace that outstrips our ability to evaluate them. Yet nearly all existing implementations adopt a single role-play paradigm in which the AI portrays a patient and the student acts as a physician, without examining whether alternative role configurations produce different teaching outcomes, or how the quality of any such agent should be judged in the first place.</p><p>These developments expose a more fundamental gap: we lack a validated way to judge whether an AI teaching agent teaches well. Existing evaluation frameworks address clinical intervention outcomes or clinical task accuracy [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref16">16</xref>] but not the teaching-specific dimensions that determine educational value, such as adaptive difficulty calibration, knowledge coverage, and formative feedback quality. In practice, quality is often inferred from the aggregate scores generated by commercial teaching platforms, yet these scores are produced by undisclosed criteria that cannot be independently verified, are not decomposable by dimension, and are not comparable across agents, leaving it unclear what they actually measure and whether they reflect the teaching quality of the agent or merely the surface characteristics of a student&#x2019;s responses. This evaluation gap has a direct consequence: without a trustworthy measure of teaching quality, the design variables that might improve these agents, including role-play configuration, content domain, and learner characteristics, such as gender, cannot be studied systematically and remain largely unexplored within a single controlled context [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Endocrinology is well suited to investigating these questions, spanning molecular mechanisms to chronic disease management within a single curricular module, yet remaining heavily reliant on traditional didactics, with simulation-based approaches underrepresented [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>]. Against this background, we deployed 8 AI teaching agents covering a complete endocrinology curriculum, each employing a distinct role-play paradigm, and analyzed 167 student-agent dialogues. Because the platform&#x2019;s built-in scores proved uninterpretable as measures of teaching quality, we developed a purpose-built 8-dimension rubric, validated against multiple independent LLM evaluators and a human expert, as a transparent external benchmark. To our knowledge, this is the first study to (1) propose and validate a multidimensional framework for evaluating AI teaching agents in medical education, (2) examine how role-play design, content domain, and learner gender relate to teaching quality within a unified curriculum, and (3) compare platform-generated scores against a transparent external standard to clarify what each actually captures. Beyond the specific agents studied, this work offers a reusable approach to a problem the field will increasingly face: how to see what an AI teaching agent is actually teaching.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Participants</title><p>This cross-sectional observational study was conducted during the autumn semester of 2025 at Guangzhou Medical University with 22 fourth-year undergraduate students majoring in basic medical sciences, who participated as part of their endocrinology curriculum. The cohort comprised 15 female and 7 male students. Each student interacted with all 8 AI teaching agents sequentially over the course of the semester. Owing to individual absences, the number of completed dialogues per agent ranged from 19 to 22, yielding 167 valid dialogues in total.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>All participants were informed of the study purpose, and the study was approved by the institutional review board of Guangzhou Medical University (202607001).</p></sec><sec id="s2-3"><title>AI Teaching Agent Design and Deployment</title><p>Eight AI teaching agents were constructed and deployed on the Chaoxing e-Learning platform (Beijing Century Chaoxing Information Technology Development Co, Ltd), a widely used commercial platform in Chinese higher education. Each agent covered 1 chapter of the endocrinology curriculum. Two chapters addressed foundational content (a general introduction and organ morphology), and 6 addressed clinical or disease-oriented content (hypothalamic-pituitary diseases, thyroid diseases, adrenal diseases, glucose metabolism disorders, lipid metabolism disorders, and calcium-phosphorus metabolism disorders).</p><p>Each agent employed a distinct role-play scenario assigned to 1 of 4 role paradigms: patient (A1, A4, A5), student (A2, A3), expert (A6, A7), and family member (A8); full configurations are provided in <xref ref-type="table" rid="table1">Table 1</xref>. The 4 paradigms were chosen to represent 4 prototypical interaction scenarios in clinical medicine (doctor and patient, teacher and student, expert consultation, and family communication), thereby sampling a range of communicative and pedagogical demands rather than instantiating a single learning theory a priori. Where relevant, the resulting behaviors could be related post hoc to established educational concepts; for example, the student paradigm elicits scaffolding within the learner&#x2019;s zone of proximal development [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. We emphasize that because each agent combined 1 role paradigm with 1 content chapter, role and content were confounded by design.</p><p>Key implementation differences among the agents lay in the assigned AI persona and the reciprocal role assigned to the student, which together determined the communicative register and the pedagogical stance of each dialogue (<xref ref-type="table" rid="table1">Table 1</xref>). For example, within the expert paradigm, agent A6 positioned the student as a learner in a mentorship model, whereas agent A7 positioned the student as a patient, providing both a physician and a patient perspective. Agent A8 was deliberately revised after the initial deployment of the first 7 agents, when their empathic engagement was judged to be limited; only its prompt was modified, while all other settings were held constant, to strengthen an empathic family-member persona. For each agent, the research team specified the instructional content, learning objectives, knowledge sources, role-play scenario, and safety and behavioral constraints; the complete per-agent configuration, including prompts, is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Overview of AI teaching agent configurations.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Agent</td><td align="left" valign="bottom">Content topic</td><td align="left" valign="bottom">Role paradigm</td><td align="left" valign="bottom">AI role</td><td align="left" valign="bottom">Student role</td></tr></thead><tbody><tr><td align="left" valign="top">A1</td><td align="left" valign="top">General introduction to endocrinology</td><td align="left" valign="top">Patient</td><td align="left" valign="top">Patient (endocrinology inpatient)</td><td align="left" valign="top">Physician (endocrinology intern)</td></tr><tr><td align="left" valign="top">A2</td><td align="left" valign="top">Organ morphology</td><td align="left" valign="top">Student</td><td align="left" valign="top">Learner (first-year medical student)</td><td align="left" valign="top">Instructor (anatomy and histology teacher)</td></tr><tr><td align="left" valign="top">A3</td><td align="left" valign="top">Hypothalamic-pituitary diseases</td><td align="left" valign="top">Student</td><td align="left" valign="top">Learner (average-performing student)</td><td align="left" valign="top">Instructor (top-performing class representative)</td></tr><tr><td align="left" valign="top">A4</td><td align="left" valign="top">Thyroid diseases</td><td align="left" valign="top">Patient</td><td align="left" valign="top">Patient (inquisitive, detail-seeking patient)</td><td align="left" valign="top">Physician (patient&#x2019;s physician)</td></tr><tr><td align="left" valign="top">A5</td><td align="left" valign="top">Adrenal diseases</td><td align="left" valign="top">Patient</td><td align="left" valign="top">Patient (Cushing syndrome patient)</td><td align="left" valign="top">Physician (endocrinologist)</td></tr><tr><td align="left" valign="top">A6</td><td align="left" valign="top">Glucose metabolism disorders</td><td align="left" valign="top">Expert</td><td align="left" valign="top">Expert (senior diabetes clinician-researcher)</td><td align="left" valign="top">Learner (basic medical science student)</td></tr><tr><td align="left" valign="top">A7</td><td align="left" valign="top">Lipid metabolism disorders</td><td align="left" valign="top">Expert</td><td align="left" valign="top">Expert (attending physician)</td><td align="left" valign="top">Patient (patient with obesity)</td></tr><tr><td align="left" valign="top">A8</td><td align="left" valign="top">Calcium-phosphorus metabolism disorders</td><td align="left" valign="top">Family</td><td align="left" valign="top">Family member (older patient with osteoporosis)</td><td align="left" valign="top">Caregiver (grandchild)</td></tr></tbody></table></table-wrap></sec><sec id="s2-4"><title>Data Collection</title><p>Two types of data were collected from each student-agent dialogue. First, the platform&#x2019;s built-in scoring system generated a single aggregate score per dialogue as an automated assessment of student performance. The platform does not disclose how these scores are generated, and its scoring mechanism could not be independently obtained or verified; only aggregate scores were available to instructors, with no dimensional breakdown. Second, complete dialogue transcripts were exported from the platform for independent external evaluation.</p></sec><sec id="s2-5"><title>Eight-Dimension Pedagogical Evaluation Rubric</title><p>To enable standardized cross-agent comparison of teaching quality, we developed an 8-dimension evaluation rubric (<xref ref-type="table" rid="table2">Table 2</xref>) informed by existing frameworks for AI conversational agents in health care [<xref ref-type="bibr" rid="ref14">14</xref>], AI agent evaluation in clinical medicine [<xref ref-type="bibr" rid="ref16">16</xref>], and educational agent effectiveness research [<xref ref-type="bibr" rid="ref17">17</xref>], supplemented by expert discussion within the research team. Unlike platform-generated scoring, the rubric was designed to assess agent teaching quality rather than student response quality. The complete scoring prompt is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>Dimension weights were assigned by the research team on conceptual grounds: medical knowledge accuracy and pedagogical guidance, as the core teaching functions, received the highest weight (20 points each), whereas student engagement elicitation and feedback quality, as supportive dimensions, received the lowest (5 points each). We note as a limitation that these weights were set without formal student or educator input, and that empirically derived weights are a direction for future work.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Eight-dimension pedagogical evaluation rubric for AI teaching agents.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimension</td><td align="left" valign="bottom">Points (total=100)</td><td align="left" valign="bottom">Scoring focus</td></tr></thead><tbody><tr><td align="left" valign="top">Medical knowledge accuracy</td><td align="left" valign="top">20</td><td align="left" valign="top">Factual correctness and absence of medical errors</td></tr><tr><td align="left" valign="top">Pedagogical guidance ability</td><td align="left" valign="top">20</td><td align="left" valign="top">Use of questioning, scaffolding, and structured guidance</td></tr><tr><td align="left" valign="top">Knowledge-point coverage efficiency</td><td align="left" valign="top">15</td><td align="left" valign="top">Breadth and completeness of curriculum content addressed</td></tr><tr><td align="left" valign="top">Role-play quality</td><td align="left" valign="top">15</td><td align="left" valign="top">Consistency, naturalness, and appropriateness of role portrayal</td></tr><tr><td align="left" valign="top">Adaptive difficulty calibration</td><td align="left" valign="top">10</td><td align="left" valign="top">Adjustment of complexity in response to student level</td></tr><tr><td align="left" valign="top">Medical safety boundary</td><td align="left" valign="top">10</td><td align="left" valign="top">Avoidance of harmful advice and appropriate referral behavior</td></tr><tr><td align="left" valign="top">Student engagement elicitation</td><td align="left" valign="top">5</td><td align="left" valign="top">Prompting of active student participation and critical thinking</td></tr><tr><td align="left" valign="top">Assessment and feedback quality</td><td align="left" valign="top">5</td><td align="left" valign="top">Specificity and constructiveness of formative feedback</td></tr></tbody></table></table-wrap></sec><sec id="s2-6"><title>Rubric-Based Scoring and Validation</title><p>All 167 dialogue transcripts were scored across the 8 dimensions using LLMs operating under standardized instructions [<xref ref-type="bibr" rid="ref15">15</xref>]. To assess the robustness and reliability of the scoring, 3 LLMs from different developers were used as evaluators: Claude Opus 4.8 (Anthropic), representing a leading international model, and DeepSeek (DeepSeek V4 Artificial Intelligence Basic Technology Research Co, Ltd) and Qwen 2.5 (Alibaba Cloud), two widely used Chinese LLMs selected for their strong Chinese-language capability given that all dialogues used Chinese. DeepSeek and Qwen were added at the recommendation of a reviewer and on the basis of our recent study of mainstream domestic and international LLMs for medical assessment tasks [<xref ref-type="bibr" rid="ref22">22</xref>].</p><p>All scoring was performed programmatically through each model&#x2019;s application programming interface under a fixed request configuration, with an intercall interval of 1.0 second and up to 3 retries per call. To promote deterministic, reproducible outputs, DeepSeek and Qwen were queried with temperatures of 0 and top-p of 0.01; Claude Opus 4.8 was queried at its default setting. Each model independently applied the identical 8-dimension rubric and scored every dialogue 4 times, and the mean of the 4 runs was used as that model&#x2019;s final score. This repeated-scoring design followed directly from our recent finding that the reliability of LLM-based scoring is better captured by repeated runs and cross-run variance than by single-run performance [<xref ref-type="bibr" rid="ref22">22</xref>]. Within-model reproducibility across the 4 runs was quantified by the SD of the repeated scores; reproducibility was high, indicating that scores were stable rather than single, unrepeatable judgments [<xref ref-type="bibr" rid="ref23">23</xref>].</p><p>To calibrate the AI-based scoring against human judgment, a medical-education expert independently rescored a stratified subset of 40 dialogues using the identical 8-dimension rubric. The subset was stratified by Claude total score to span high, medium, and low quality and to cover all 8 agents, and the expert was blinded to the AI scores. Agreement between AI and expert ratings was quantified by the Pearson correlation and the intraclass correlation coefficient (ICC), computed at the total-score level and per dimension [<xref ref-type="bibr" rid="ref24">24</xref>]. Because rescoring was performed by a single expert, expert-to-expert reliability could not be computed; this is addressed in the Limitations section.</p></sec><sec id="s2-7"><title>Statistical Analysis</title><p>Given nonnormal score distributions and heterogeneous variances across agents (Shapiro-Wilk and Levene tests), nonparametric methods were used throughout. Platform-generated scores were compared across agents using the Kruskal-Wallis <italic>H</italic> test, with post hoc Dunn tests and Bonferroni correction for pairwise comparisons. Effect sizes were reported as &#x03B7;<sup>2</sup> for omnibus tests and Cohen <italic>d</italic> for selected contrasts. Between-agent variation in each rubric dimension was summarized using the coefficient of variation (CV).</p><p>Comparisons across role paradigms were treated as descriptive, because the independent unit was the agent and each role group comprised only 1 to 3 agents; formal inferential testing was therefore not performed at the agent level. The divergence between platform-based and rubric-based agent rankings was characterized through concrete rank comparisons rather than a single agent-level correlation coefficient, which would be unstable at this sample size. Learner gender differences were examined across dimensions using Mann-Whitney <italic>U</italic> tests, with results interpreted in light of the small and imbalanced sample. The correspondence between rubric-based agent quality and students&#x2019; end-of-course chapter examination performance was examined descriptively. All analyses were conducted in Python (version 3.11), with a 2-sided significance threshold of <italic>P</italic>&#x003C;.05.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Platform-Generated Scores Varied Across Agents</title><p>Chaoxing platform&#x2013;generated scores from 167 student-agent dialogues differed significantly across the 8 agents (Kruskal-Wallis <italic>H</italic><sub>7</sub>=57.981; <italic>P</italic>&#x003C;.001; &#x03B7;<sup>2</sup>=0.321; <xref ref-type="fig" rid="figure1">Figure 1</xref>), with 9 significant pairwise differences identified by post hoc Dunn tests (Bonferroni-corrected <italic>P</italic>&#x003C;.05). Basic science agents (A2, A3) received the highest scores, while clinical disease agents (A4, A5, A7) received the lowest. While these differences indicate that agent design influenced student performance under platform scoring, the platform does not disclose how its scores are generated, and its scoring mechanism could not be independently verified, precluding any interpretation of what drove them or whether they reflect genuine agent teaching quality. Therefore, the 8-dimension rubric was applied as an independent external audit, reported in the following sections.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Distribution of Chaoxing platform&#x2013;generated scores across 8 agents (22 students; 167 valid dialogues). Box plots show median, IQR, and 1.5&#x00D7;IQR whiskers; individual scores are jittered dots; outliers (&#x003E;1.5&#x00D7;IQR) are in orange. Blue numerals above boxes denote means. Agents not sharing a common letter (a vs b) differ significantly ( Bonferroni-corrected <italic>P</italic>&#x003C;.05).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e96819_fig01.png"/></fig></sec><sec id="s3-2"><title>LLM Evaluators Differed in Scoring Discrimination</title><p>To evaluate the robustness of the rubric-based scoring, the 8-dimension rubric was independently applied by 3 LLM evaluators from different developers (Claude, DeepSeek, and Qwen), each scoring all 167 dialogues 4 times, with high within-model reproducibility (<xref ref-type="table" rid="table3">Table 3</xref>; mean within-model SD&#x2264;0.36).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Comparison of 3 LLM evaluators on the 8-dimension rubric.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Claude (Opus 4.8)</td><td align="left" valign="bottom">DeepSeek V4</td><td align="left" valign="bottom">Qwen 2.5</td></tr></thead><tbody><tr><td align="left" valign="top">Dialogues scored (&#x00D7;4 runs), n</td><td align="left" valign="top">167</td><td align="left" valign="top">167</td><td align="left" valign="top">167</td></tr><tr><td align="left" valign="top">Intrarater SD (score stability)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="top">0.36</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.17</td></tr><tr><td align="left" valign="top">Agent-total score range</td><td align="left" valign="top">61&#x2010;92</td><td align="left" valign="top">82&#x2010;95</td><td align="left" valign="top">75&#x2010;100</td></tr><tr><td align="left" valign="top">Agent-total SD (discrimination)<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">11.6</td><td align="left" valign="top">4.0</td><td align="left" valign="top">7.3</td></tr><tr><td align="left" valign="top">Overall ceiling rate<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup>, %</td><td align="left" valign="top">17</td><td align="left" valign="top">30</td><td align="left" valign="top">51</td></tr><tr><td align="left" valign="top">Mean achievement rate, %</td><td align="left" valign="top">82</td><td align="left" valign="top">90</td><td align="left" valign="top">92</td></tr><tr><td align="left" valign="top">Rank correlation with Claude (&#x03C1;)<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.71</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Intrarater SD: mean within-model SD across the 4 runs (lower=more stable).</p></fn><fn id="table3fn2"><p><sup>b</sup>Agent-total SD: SD of the 8 agent-level totals (higher=better discrimination).</p></fn><fn id="table3fn3"><p><sup>c</sup>Overall ceiling rate: proportion of dialogue&#x00D7;dimension cells scored at the maximum (higher=stronger ceiling effect).</p></fn><fn id="table3fn4"><p><sup>d</sup>Rank correlation: Spearman &#x03C1; of agent rankings with Claude (both <italic>P</italic>&#x003C;.05).</p></fn><fn id="table3fn5"><p><sup>e</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>The 3 models agreed broadly on the relative ranking of agents (Spearman &#x03C1;=0.64 and 0.71 for Claude vs DeepSeek and Qwen) but differed substantially in discrimination. DeepSeek and Qwen showed pronounced leniency, scoring large proportions of dialogues at the ceiling (<xref ref-type="fig" rid="figure2">Figure 2</xref>; Qwen: above 90% on medical safety boundary, student engagement elicitation, and assessment and feedback quality; DeepSeek: 60-77% on the same dimensions). This compressed between-agent differences, yielding narrow score ranges (82&#x2010;95 and 75&#x2010;100) and low discrimination (SD of agent total scores was 4.0 and 7.3). Claude showed a minimal ceiling effects except on the two 5-point dimensions (student engagement elicitation, assessment and feedback quality), with the widest range (61-92) and greatest discrimination (SD 11.6).</p><p>Because the analysis aimed to resolve differences in teaching quality among agents, an evaluator able to discriminate between them was required; Claude was therefore used as the primary evaluator, with its agreement against human expert judgment examined separately. The leniency of the other evaluators is itself a relevant finding, indicating that ceiling effects should be anticipated when LLMs serve as evaluators of educational quality.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Ceiling effects across the 3 large language model evaluators on the 8 scoring dimensions. Each cell shows the proportion of the 167 dialogues scored at the maximum for that dimension; darker red indicates a stronger ceiling effect (less discrimination). ADC: adaptive difficulty calibration; AFQ: assessment and feedback quality; KCE: knowledge-point coverage efficiency; MKA: medical knowledge accuracy; MSB: medical safety boundary; PGA: pedagogical guidance ability; RPQ: role-play quality; SEE: student engagement elicitation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e96819_fig02.png"/></fig></sec><sec id="s3-3"><title>Agents Differed Mainly in Knowledge, Not in Role Performance</title><p>Applying the 8-dimension rubric, agent total scores ranged widely (60.6&#x2010;91.6; <xref ref-type="fig" rid="figure3">Figure 3</xref>), with A6 and A3 scoring highest and A7 and A8 lowest. This variation was not uniform across dimensions. It was largest on the 2 knowledge-related dimensions, namely knowledge-point coverage (CV=27.3%) and medical knowledge accuracy (CV=18.5%), indicating that agents differed most in what and how accurately they taught. In contrast, role-play quality was the most uniform dimension (CV=5.7%), with all agents sustaining their assigned personas comparably well. Adaptive difficulty control was low across nearly all agents, emerging as a shared weakness rather than a source of between-agent variation. Thus, the agents were broadly comparable in role enactment but diverged mainly in the knowledge content they delivered.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Eight-dimension rubric profiles of the 8 AI teaching agents. Scores are the mean of 4 independent runs by Claude Opus 4.8. Bubble size indicates the dimension's maximum score, color indicates the achievement rate; and number indicates the raw score. ADC: adaptive difficulty calibration; AFQ: assessment and feedback quality; KCE: knowledge-point coverage efficiency; MKA: medical knowledge accuracy; MSB: medical safety boundary; PGA: pedagogical guidance ability; RPQ: role-play quality; SEE: student engagement elicitation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e96819_fig03.png"/></fig></sec><sec id="s3-4"><title>Role Paradigms Shaped Dimension Profiles, but Empathy Did Not Ensure Knowledge Coverage</title><p>Grouping agents by role paradigm revealed distinct dimension profiles (<xref ref-type="fig" rid="figure4">Figure 4</xref>). At the group level, student-type agents showed the strongest pedagogical guidance (group-mean pedagogical guidance ability was 17.7), whereas expert-type agents showed the lowest medical safety (group-mean medical safety boundary was 6.7). The family-member agent illustrated a notable dissociation between empathy and knowledge: it attained one of the highest role-play scores (role-play quality=14.1) yet the lowest knowledge-point coverage of all agents (knowledge-point coverage efficiency=5.3) and the lowest total score (60.6). Notably, this agent had been deliberately revised after the first 7 agents showed generally limited empathic engagement in student use; only its prompt was adjusted&#x2014;while all other settings were held constant&#x2014;to strengthen the empathic, family-member persona. Despite this targeted enhancement of empathy, it remained weakest in knowledge coverage, indicating that strengthening empathy did not translate into broader knowledge delivery. How to balance affective engagement and knowledge coverage within a single agent thus remains an open question. Because each role group comprised only 2 to 3 agents, these between-group comparisons are descriptive; formal inferential testing was not performed at the agent level.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Eight-dimension score distributions across the 4 role paradigms: patient (A1, A4, A5; n=64 dialogues), student (A2, A3; n=44), expert (A6, A7; n=40), and family member (A8; n=19). Each panel shows 1 dimension (scored out of its maximum, indicated in parentheses). Boxes show median and IQR; points are individual dialogue scores. Between-group differences are descriptive; formal inferential testing was not performed at the agent level. ADC: adaptive difficulty calibration; AFQ: assessment and feedback quality; KCE: knowledge-point coverage efficiency; MKA: medical knowledge accuracy; MSB: medical safety boundary; PGA: pedagogical guidance ability; RPQ: role-play quality; SEE: student engagement elicitation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e96819_fig04.png"/></fig></sec><sec id="s3-5"><title>Student Gender Showed No Detectable Effect on Teaching Quality Scores</title><p>Male (n=7) and female (n=15) students were compared across all 8 dimensions using Mann-Whitney <italic>U</italic> tests. No significant difference was detected in any dimension or in total score (79.1 vs 80.1; <italic>P</italic>=.82; all dimension-level <italic>P</italic>&#x003E;.05; Cohen <italic>d</italic>=0.07 for total score). Given the small, imbalanced sample and the very small observed effect size, this analysis was substantially underpowered; the absence of a detectable difference should therefore not be interpreted as evidence of gender-equitable delivery, as a type II error cannot be excluded.</p></sec><sec id="s3-6"><title>AI Scores Agreed With Expert Ratings on Overall and Cognitive-Process Dimensions</title><p>To validate the rubric-based AI scoring, 40 dialogues spanning all 8 agents and the full quality range were independently rescored by a medical-education expert using an identical rubric, blinded to the AI scores (<xref ref-type="fig" rid="figure5">Figure 5A</xref>). AI and expert ratings were closely aligned in magnitude (mean totals 79.1 vs 78.5) and were positively correlated at the total-score level (<italic>r</italic>=0.61; ICC=0.51). Because the two derive from a highly consistent automated rater and a single subjective human expert, close but not exact agreement was expected; the aim was to confirm convergence on the same construct, not identity.</p><p>Agreement was strongest on the cognitive-process dimensions, ranging from moderate for knowledge accuracy to highest for adaptive difficulty control. For role-play quality and the two 5-point dimensions, the ICC was near zero or negative, but this reflects a ceiling effect rather than rater disagreement: both AI and expert rated these dimensions uniformly high (for role-play quality, 13.4 vs 13.3 out of 15), leaving minimal between-agent variance (<xref ref-type="fig" rid="figure5">Figure 5B</xref>). Because the ICC indexes relative discrimination, it is driven toward zero when there is little variance to rank, and its low value here does not indicate that the raters disagreed; on the contrary, they consistently converged on high scores [<xref ref-type="bibr" rid="ref24">24</xref>]. The moderate agreement on knowledge accuracy likewise does not undermine the study&#x2019;s key knowledge-related finding, which rests on a rank-level difference too large to depend on rating precision: A8 scored markedly lowest on knowledge coverage under both AI and expert scoring.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Agreement between Claude AI and expert ratings on 40 dialogues, independently scored by a medical-education expert using the same 8-dimension rubric. (A) Total score agreement: expert versus Claude total scores (the dashed line indicates perfect agreement). (B) Per-dimension agreement represented by intraclass correlation coefficient (ICC) per dimension (green&#x2265;0.5; amber 0.25-0.5; red&#x003C;0.25). ADC: adaptive difficulty calibration; AFQ: assessment and feedback quality; KCE: knowledge-point coverage efficiency; MKA: medical knowledge accuracy; MSB: medical safety boundary; PGA: pedagogical guidance ability; RPQ: role-play quality; SEE: student engagement elicitation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e96819_fig05.png"/></fig></sec><sec id="s3-7"><title>Platform and Examination Performance Captured Constructs Distinct From Rubric-Based Agent Quality</title><p>Agent rankings from platform scores and from the 8-dimension rubric diverged substantially (<xref ref-type="fig" rid="figure6">Figure 6A</xref>). Two agents illustrate this most clearly: A8 ranked third by platform score but last by rubric quality, and A6 rose from fourth to first. Rather than indicating that either measure is invalid, this divergence reflects that the two capture different constructs&#x2014;platform scores index student performance during the interaction, whereas the rubric indexes the teaching quality of the agent itself. The two are therefore complementary rather than interchangeable.</p><p>There was a similar lack of association between rubric quality and end-of-course chapter examination performance (<xref ref-type="fig" rid="figure6">Figure 6</xref>B): higher agent quality did not correspond to higher chapter examination scores (eg, A3 showed high quality but the lowest examination score rate, whereas A7 and A8 showed low overall quality but high examination scores). This absence of a systematic association should be interpreted cautiously, given the small number of agents, the near-ceiling examination scores, and the lack of individual baseline data; it indicates only that agent teaching quality and this particular examination did not track one another, not that better teaching is ineffective.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Platform and examination performance versus rubric-based agent quality. (A) Rank divergence: agent rankings by platform score (student performance) vs by 8-dimension quality (agent quality); 1=best. (B) Eight-dimension agent quality versus end-of-course chapter examination score rate.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e96819_fig06.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>This study set out to test whether platform-generated scores, the default metric for AI teaching agents in many educational platforms, adequately capture teaching quality, and to ask what a transparent, multidimensional alternative would reveal. The findings speak less to any single agent than to how the field should evaluate, and design, conversational AI tutors.</p><p>Evaluation is not a solved problem, and the default metric may mislead. The most consequential implication is that the metric an educator happens to have at hand can point in the wrong direction. When a platform score and a criterion-based rubric rank the same agents in nearly opposite orders, the choice of metric is not a technical detail but a determinant of which agents get adopted, refined, or discarded. Crucially, this does not make either metric wrong. A platform score that reflects student performance during an interaction and a rubric that reflects the agent&#x2019;s teaching behavior are answering different questions; the problem arises only when one is silently substituted for the other. As AI tutors proliferate faster than the frameworks to evaluate them [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref14">14</xref>], this argues for treating evaluation itself as a first-class research object, reporting what a metric measures, and pairing convenient platform analytics with transparent, education-specific criteria rather than trusting either alone [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Automated evaluation is promising but must be bounded. Using LLMs to score pedagogical quality is attractive because it scales, but our results caution against naive trust [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Independently developed models converged on the ordering of agents yet differed markedly in leniency, showing that agreement on ranking can coexist with disagreement on absolute standards, and that some models are too lenient to discriminate at all [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Human calibration therefore remains indispensable [<xref ref-type="bibr" rid="ref27">27</xref>], and it is not uniform across constructs: automated and expert judgments aligned well on cognitive-process dimensions but poorly on more subjective, affective ones [<xref ref-type="bibr" rid="ref27">27</xref>]. This suggests a pragmatic division of labor, in which automated scoring is trusted most for the dimensions where it is demonstrably calibrated, while human judgment is reserved for those where meaning is contested [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>Designing for the whole of teaching, not its most visible parts, and comparing role paradigms, rather than the field&#x2019;s default patient-doctor script [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>], exposed a design tension that is easy to miss when engagement is the headline goal. An agent optimized to be more empathic became more engaging without becoming more knowledgeable, a reminder that affective and cognitive teaching functions do not automatically move together [<xref ref-type="bibr" rid="ref17">17</xref>], and that improving the salient, likable qualities of a tutor can leave its instructional substance untouched. Rather than seeking a single best role, this points toward designing for complementarity, orchestrating different pedagogical strengths across an interaction [<xref ref-type="bibr" rid="ref16">16</xref>], and toward evaluation that can detect when a more engaging agent is not in fact teaching more. The most uniform finding across every agent, namely persistent weakness in adapting difficulty and providing formative feedback, reinforces the same lesson: the harder and less visible pedagogical skills are precisely those that current prompt-based designs deliver least well [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref28">28</xref>].</p><p>Future directions should consider 3 points. First, disentangling role from content requires a factorial design that crosses role paradigms with content domains, so that the effect of how an agent teaches can be separated from what it teaches [<xref ref-type="bibr" rid="ref5">5</xref>]. Second, establishing educational value, rather than teaching quality alone, requires learning-outcome designs with pre- and postlearning knowledge assessment and individual-level linkage [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref19">19</xref>], ideally using more discriminating assessments than a near-ceiling course examination. Third, the persistent deficits in adaptive difficulty and feedback motivate designs that move beyond static prompting toward dynamic learner modeling and memory-augmented tutoring [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>], evaluated against the dimensions on which current agents underperform. More broadly, the 8-dimension evaluation framework introduced here offers a transferable reference for the development of future teaching agents, and of medical teaching agents in particular, by making explicit the pedagogical dimensions against which such agents should be designed and assessed.</p><p>Undeniably, this is a small, single-institution study in one discipline, and its design confounds role paradigm with content domain, each agent being one role and one chapter, so role-related observations are descriptive rather than causal and no independent content-domain effect can be claimed. Because role groups contain few agents, group comparisons are descriptive and were not tested inferentially at the agent level. Expert calibration rested on a single expert, precluding expert-to-expert reliability, and was weaker for subjective dimensions. The gender comparison was underpowered and cannot support claims of equitable delivery, the rubric weights were team-assigned without formal stakeholder input, and teaching quality rather than learning gain was the outcome assessed. These constraints frame the study as an exploratory, framework-building effort whose specific estimates await confirmation in larger, factorial, outcome-linked designs.</p><p>Platform scores answer a different question than teaching quality and should complement rather than replace criterion-based evaluation. By making the dimensions of teaching explicit, and by validating those judgments against expert and cross-model agreement, a multidimensional rubric reveals distinctions that convenient metrics miss, including where more engaging agents are not more instructive and where all current agents falter. Beyond any single result, the study suggests that how we evaluate AI teaching agents is inseparable from how we improve them.</p></sec></body><back><ack><p>The authors declare that generative AI was used in the preparation of this manuscript. Claude Opus 4.8 was employed to enhance the language and readability of the draft. The final version was thoroughly reviewed, revised, and approved by the authors, who take full responsibility for the content.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the following grants: the 2026 Educational Science Planning and Teaching Reform Projects of Guangzhou Medical University (2026JXGG02) and the 2025 Guangdong Provincial Graduate Education Innovation Program (2025JGXM_139).</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during the current study are available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CV</term><def><p>coefficient of variation</p></def></def-item><def-item><term id="abb2">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Long</surname><given-names>C</given-names> </name><name name-style="western"><surname>Tsai</surname><given-names>K</given-names> </name><name name-style="western"><surname>Napolitano</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Khawaja</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Leung</surname><given-names>AM</given-names> </name></person-group><article-title>The current status of preclinical endocrine education in U.S. medical schools</article-title><source>Endocr Pract</source><year>2022</year><month>08</month><volume>28</volume><issue>8</issue><fpage>744</fpage><lpage>748</lpage><pub-id pub-id-type="doi">10.1016/j.eprac.2022.04.008</pub-id><pub-id pub-id-type="medline">35452814</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Han</surname><given-names>YT</given-names> </name><name name-style="western"><surname>Lyu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>LM</given-names> </name></person-group><article-title>Advantages and challenges of applying artificial intelligence in medical education</article-title><source>Zhonghua Liu Xing Bing Xue Za Zhi</source><year>2026</year><month>02</month><day>10</day><volume>47</volume><issue>2</issue><fpage>200</fpage><lpage>206</lpage><pub-id pub-id-type="doi">10.3760/cma.j.cn112338-20250512-00309</pub-id><pub-id pub-id-type="medline">41765658</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Enhancing medical education with chatbots: a randomized controlled trial on standardized patients for colorectal cancer</article-title><source>BMC Med Educ</source><year>2024</year><month>12</month><day>20</day><volume>24</volume><issue>1</issue><fpage>1511</fpage><pub-id pub-id-type="doi">10.1186/s12909-024-06530-8</pub-id><pub-id pub-id-type="medline">39707245</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>&#x00D6;nc&#x00FC;</surname><given-names>S</given-names> </name><name name-style="western"><surname>Torun</surname><given-names>F</given-names> </name><name name-style="western"><surname>&#x00DC;lk&#x00FC;</surname><given-names>HH</given-names> </name></person-group><article-title>AI-powered standardised patients: evaluating ChatGPT-4o&#x2019;s impact on clinical case management in intern physicians</article-title><source>BMC Med Educ</source><year>2025</year><month>02</month><day>20</day><volume>25</volume><issue>1</issue><fpage>278</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-06877-6</pub-id><pub-id pub-id-type="medline">39979969</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xiao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>X</given-names> </name></person-group><article-title>Is the use of standardized patients more effective than role-playing in medical education? A meta-analysis</article-title><source>Front Med (Lausanne)</source><year>2025</year><month>06</month><day>18</day><volume>12</volume><fpage>1601116</fpage><pub-id pub-id-type="doi">10.3389/fmed.2025.1601116</pub-id><pub-id pub-id-type="medline">40606465</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Chuang</surname><given-names>CL</given-names> </name><etal/></person-group><article-title>Role-play of real patients improves the clinical performance of medical students</article-title><source>J Chin Med Assoc</source><year>2021</year><month>02</month><day>1</day><volume>84</volume><issue>2</issue><fpage>183</fpage><lpage>190</lpage><pub-id pub-id-type="doi">10.1097/JCMA.0000000000000431</pub-id><pub-id pub-id-type="medline">32925298</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Flanagan</surname><given-names>OL</given-names> </name><name name-style="western"><surname>Cummings</surname><given-names>KM</given-names> </name></person-group><article-title>Standardized patients in medical education: a review of the literature</article-title><source>Cureus</source><year>2023</year><month>07</month><day>17</day><volume>15</volume><issue>7</issue><fpage>e42027</fpage><pub-id pub-id-type="doi">10.7759/cureus.42027</pub-id><pub-id pub-id-type="medline">37593270</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Simulated patient systems powered by large language model-based AI agents offer potential for transforming medical education</article-title><source>Commun Med (Lond)</source><year>2025</year><month>12</month><day>19</day><volume>6</volume><issue>1</issue><fpage>27</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-01283-x</pub-id><pub-id pub-id-type="medline">41420084</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Br&#x00FC;gge</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ricchizzi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Arenbeck</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Large language models improve clinical decision making of medical students through patient simulation and structured feedback: a randomized controlled trial</article-title><source>BMC Med Educ</source><year>2024</year><month>11</month><day>28</day><volume>24</volume><issue>1</issue><fpage>1391</fpage><pub-id pub-id-type="doi">10.1186/s12909-024-06399-7</pub-id><pub-id pub-id-type="medline">39609823</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Effectiveness of AI-assisted medical education for Chinese undergraduate medical students: a meta-analysis</article-title><source>BMC Med Educ</source><year>2025</year><month>08</month><day>27</day><volume>25</volume><issue>1</issue><fpage>1207</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-07770-y</pub-id><pub-id pub-id-type="medline">40866973</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aster</surname><given-names>A</given-names> </name><name name-style="western"><surname>Laupichler</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Rockwell-Kollmann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Masala</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bala</surname><given-names>E</given-names> </name><name name-style="western"><surname>Raupach</surname><given-names>T</given-names> </name></person-group><article-title>ChatGPT and other large language models in medical education - scoping literature review</article-title><source>Med Sci Educ</source><year>2024</year><month>11</month><day>13</day><volume>35</volume><issue>1</issue><fpage>555</fpage><lpage>567</lpage><pub-id pub-id-type="doi">10.1007/s40670-024-02206-6</pub-id><pub-id pub-id-type="medline">40144083</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vrdoljak</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boban</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Vilovi&#x0107;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kumri&#x0107;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bo&#x017E;i&#x0107;</surname><given-names>J</given-names> </name></person-group><article-title>A review of large language models in medical education, clinical decision support, and healthcare administration</article-title><source>Healthcare (Basel)</source><year>2025</year><month>03</month><day>10</day><volume>13</volume><issue>6</issue><fpage>603</fpage><pub-id pub-id-type="doi">10.3390/healthcare13060603</pub-id><pub-id pub-id-type="medline">40150453</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elhilali</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ngo</surname><given-names>ASH</given-names> </name><name name-style="western"><surname>Reichenpfader</surname><given-names>D</given-names> </name><name name-style="western"><surname>Denecke</surname><given-names>K</given-names> </name></person-group><article-title>Large language model-based patient simulation to foster communication skills in health care professionals: user-centered development and usability study</article-title><source>JMIR Med Educ</source><year>2025</year><month>12</month><day>12</day><volume>11</volume><fpage>e81271</fpage><pub-id pub-id-type="doi">10.2196/81271</pub-id><pub-id pub-id-type="medline">41385781</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ding</surname><given-names>H</given-names> </name><name name-style="western"><surname>Simmich</surname><given-names>J</given-names> </name><name name-style="western"><surname>Vaezipour</surname><given-names>A</given-names> </name><name name-style="western"><surname>Andrews</surname><given-names>N</given-names> </name><name name-style="western"><surname>Russell</surname><given-names>T</given-names> </name></person-group><article-title>Evaluation framework for conversational agents with artificial intelligence in health interventions: a systematic scoping review</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>02</month><day>16</day><volume>31</volume><issue>3</issue><fpage>746</fpage><lpage>761</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad222</pub-id><pub-id pub-id-type="medline">38070173</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A survey on LLM-as-a-judge</article-title><source>Innovation (Camb)</source><year>2026</year><month>06</month><day>1</day><volume>7</volume><issue>6</issue><fpage>101253</fpage><pub-id pub-id-type="doi">10.1016/j.xinn.2025.101253</pub-id><pub-id pub-id-type="medline">42254963</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gorenshtein</surname><given-names>A</given-names> </name><name name-style="western"><surname>Omar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Glicksberg</surname><given-names>BS</given-names> </name><name name-style="western"><surname>Nadkarni</surname><given-names>GN</given-names> </name><name name-style="western"><surname>Klang</surname><given-names>E</given-names> </name></person-group><article-title>AI agents in clinical medicine: a systematic review</article-title><source>medRxiv</source><year>2025</year><month>08</month><day>26</day><fpage>2025.08.22.25334232</fpage><pub-id pub-id-type="doi">10.1101/2025.08.22.25334232</pub-id><pub-id pub-id-type="medline">40909853</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name></person-group><article-title>Impact of educational agents on student&#x2019;s learning outcomes: a meta-analysis</article-title><source>Front Psychol</source><year>2026</year><month>02</month><day>24</day><volume>17</volume><fpage>1707196</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2026.1707196</pub-id><pub-id pub-id-type="medline">41815245</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alshammri</surname><given-names>F</given-names> </name><name name-style="western"><surname>Abdulshakour</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Pediatric endocrinology education among trainees: a scoping review</article-title><source>Clin Teach</source><year>2025</year><month>02</month><volume>22</volume><issue>1</issue><fpage>e70011</fpage><pub-id pub-id-type="doi">10.1111/tct.70011</pub-id><pub-id pub-id-type="medline">39743233</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Myers</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Bender</surname><given-names>NR</given-names> </name><name name-style="western"><surname>Seidel</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Weinstock</surname><given-names>RS</given-names> </name></person-group><article-title>Diabetes SPECIAL (Students Providing Education on Chronic Illness and Lifestyle): a novel preclinical medical student elective</article-title><source>Perspect Med Educ</source><year>2021</year><month>10</month><volume>10</volume><issue>5</issue><fpage>312</fpage><lpage>315</lpage><pub-id pub-id-type="doi">10.1007/s40037-020-00641-w</pub-id><pub-id pub-id-type="medline">33349906</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Daemicke</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Galt</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Samonds</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Bergan-Roller</surname><given-names>HE</given-names> </name></person-group><article-title>Challenging endocrinology students with a critical-thinking workbook</article-title><source>Adv Physiol Educ</source><year>2020</year><month>03</month><day>1</day><volume>44</volume><issue>1</issue><fpage>72</fpage><lpage>79</lpage><pub-id pub-id-type="doi">10.1152/advan.00101.2019</pub-id><pub-id pub-id-type="medline">32057267</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hodge</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Potter</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Helsby</surname><given-names>CJ</given-names> </name></person-group><article-title>Scaffolding simulation activities for medical students learning cardiopulmonary assessment: a retrospective study</article-title><source>Cureus</source><year>2025</year><month>04</month><day>10</day><volume>17</volume><issue>4</issue><fpage>e82013</fpage><pub-id pub-id-type="doi">10.7759/cureus.82013</pub-id><pub-id pub-id-type="medline">40351920</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Qu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Beyond accuracy: evaluating the reliability of large language models for medical assessment</article-title><source>Front Artif Intell</source><year>2026</year><month>07</month><day>8</day><volume>9</volume><fpage>1832829</fpage><pub-id pub-id-type="doi">10.3389/frai.2026.1832829</pub-id><pub-id pub-id-type="medline">42488377</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thomas</surname><given-names>LDW</given-names> </name><name name-style="western"><surname>Romasanta</surname><given-names>AKG</given-names> </name><name name-style="western"><surname>Pujol Priego</surname><given-names>L</given-names> </name></person-group><article-title>Jagged competencies: measuring the reliability of generative AI in academic research</article-title><source>J Bus Res</source><year>2026</year><month>01</month><volume>203</volume><fpage>115804</fpage><pub-id pub-id-type="doi">10.1016/j.jbusres.2025.115804</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mehta</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bastero-Caballero</surname><given-names>RF</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Performance of intraclass correlation coefficient (ICC) as a reliability index under various distributions in scale reliability studies</article-title><source>Stat Med</source><year>2018</year><month>08</month><day>15</day><volume>37</volume><issue>18</issue><fpage>2734</fpage><lpage>2752</lpage><pub-id pub-id-type="doi">10.1002/sim.7679</pub-id><pub-id pub-id-type="medline">29707825</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Croxford</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>First</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Evaluating clinical AI summaries with large language models as judges</article-title><source>NPJ Digit Med</source><year>2025</year><month>11</month><day>5</day><volume>8</volume><issue>1</issue><fpage>640</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02005-2</pub-id><pub-id pub-id-type="medline">41193667</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhai</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cowan</surname><given-names>AJ</given-names> </name><etal/></person-group><article-title>AI for evidence-based treatment recommendation in oncology: a blinded evaluation of large language models and agentic workflows</article-title><source>Front Artif Intell</source><year>2025</year><month>12</month><day>9</day><volume>8</volume><fpage>1683322</fpage><pub-id pub-id-type="doi">10.3389/frai.2025.1683322</pub-id><pub-id pub-id-type="medline">41446897</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Teckwani</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>AHP</given-names> </name><name name-style="western"><surname>Luke</surname><given-names>NV</given-names> </name><name name-style="western"><surname>Low</surname><given-names>ICC</given-names> </name></person-group><article-title>Accuracy and reliability of large language models in assessing learning outcomes achievement across cognitive domains</article-title><source>Adv Physiol Educ</source><year>2024</year><month>12</month><day>1</day><volume>48</volume><issue>4</issue><fpage>904</fpage><lpage>914</lpage><pub-id pub-id-type="doi">10.1152/advan.00137.2024</pub-id><pub-id pub-id-type="medline">39514712</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bogenschutz</surname><given-names>K</given-names> </name><name name-style="western"><surname>Demeter</surname><given-names>J</given-names> </name><name name-style="western"><surname>Knoderer</surname><given-names>CA</given-names> </name></person-group><article-title>Artificial intelligence in medical education: promoting active learning with a customized chatbot tool</article-title><source>J Physician Assist Educ</source><year>2026</year><month>06</month><day>1</day><volume>37</volume><issue>2</issue><fpage>216</fpage><lpage>220</lpage><pub-id pub-id-type="doi">10.1097/JPA.0000000000000742</pub-id><pub-id pub-id-type="medline">41603621</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Complete configuration of the 8 AI teaching agents.</p><media xlink:href="mededu_v12i1e96819_app1.pdf" xlink:title="PDF File, 198 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>The complete 8-dimension scoring prompt used by all evaluators.</p><media xlink:href="mededu_v12i1e96819_app2.pdf" xlink:title="PDF File, 123 KB"/></supplementary-material></app-group></back></article>