<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JME</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id>
      <journal-title>JMIR Medical Education</journal-title>
      <issn pub-type="epub">2369-3762</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v12i1e92486</article-id>
      <article-id pub-id-type="pmid">42714021</article-id>
      <article-id pub-id-type="doi">10.2196/92486</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Effect of Large Language Model–Powered Virtual Standardized Patients on History-Taking Among Undergraduate Medical Students: Propensity-Matched Cohort Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Stone</surname>
            <given-names>Alicia</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Vicente</surname>
            <given-names>Maria Asuncion</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Xia</surname>
            <given-names>Oudong</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>He</surname>
            <given-names>Yuchen</given-names>
          </name>
          <degrees>MM</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0002-5943-2978</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Chen</surname>
            <given-names>Chen</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <xref rid="aff02" ref-type="aff">2</xref>
          <xref rid="aff03" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1106-6649</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Yin</surname>
            <given-names>Rong</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff04" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-4943-1533</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Zhang</surname>
            <given-names>Wuyang</given-names>
          </name>
          <degrees>MTI</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-3684-6866</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Chang</surname>
            <given-names>Haoli</given-names>
          </name>
          <degrees>MM</degrees>
          <xref rid="aff05" ref-type="aff">5</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0001-1486-0453</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Yang</surname>
            <given-names>Wei</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff06" ref-type="aff">6</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0002-1438-567X</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Li</surname>
            <given-names>Fei</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff03" ref-type="aff">3</xref>
          <xref rid="aff07" ref-type="aff">7</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-2049-8814</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author">
          <name name-style="western">
            <surname>Li</surname>
            <given-names>Xinhua</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-8048-963X</ext-link>
        </contrib>
        <contrib id="contrib9" contrib-type="author">
          <name name-style="western">
            <surname>Xia</surname>
            <given-names>Zhuying</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff05" ref-type="aff">5</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1466-2067</ext-link>
        </contrib>
        <contrib id="contrib10" contrib-type="author">
          <name name-style="western">
            <surname>Xie</surname>
            <given-names>Xiaoyun</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff08" ref-type="aff">8</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0002-1598-5802</ext-link>
        </contrib>
        <contrib id="contrib11" contrib-type="author">
          <name name-style="western">
            <surname>Huang</surname>
            <given-names>Jing</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff08" ref-type="aff">8</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-0103-1226</ext-link>
        </contrib>
        <contrib id="contrib12" contrib-type="author">
          <name name-style="western">
            <surname>Zeng</surname>
            <given-names>Qiuming</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff03" ref-type="aff">3</xref>
          <xref rid="aff09" ref-type="aff">9</xref>
          <xref rid="aff10" ref-type="aff">10</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-2021-1484</ext-link>
        </contrib>
        <contrib id="contrib13" contrib-type="author">
          <name name-style="western">
            <surname>Yang</surname>
            <given-names>Guang</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff11" ref-type="aff">11</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-2128-2073</ext-link>
        </contrib>
        <contrib id="contrib14" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Wu</surname>
            <given-names>Jing</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <xref rid="aff03" ref-type="aff">3</xref>
          <xref rid="aff05" ref-type="aff">5</xref>
          <address>
            <institution>Department of Endocrinology</institution>
            <institution>Xiangya Hospital Central South University</institution>
            <addr-line>No.87, Xiangya road</addr-line>
            <addr-line>Kaifu District</addr-line>
            <addr-line>Changsha, Hunan, 410008</addr-line>
            <country>China</country>
            <phone>86 13574120508</phone>
            <email>wujing0731@csu.edu.cn</email>
          </address>
          <xref rid="aff12" ref-type="aff">12</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1554-9162</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff01">
        <label>1</label>
        <institution>Clinical Skills Training Center</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff02">
        <label>2</label>
        <institution>Department of Nephrology</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff03">
        <label>3</label>
        <institution>National Clinical Research Center of Geriatric Disorders</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff04">
        <label>4</label>
        <institution>Teaching and Research Section of Diagnostics</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff05">
        <label>5</label>
        <institution>Department of Endocrinology</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff06">
        <label>6</label>
        <institution>Department of Respiratory Medicine</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff07">
        <label>7</label>
        <institution>Department of Geriatric Medicine</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff08">
        <label>8</label>
        <institution>Department of Rheumatology and Immunology</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff09">
        <label>9</label>
        <institution>Department of Neurology</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff10">
        <label>10</label>
        <institution>Clinical Research Center for Neuroimmune and Neuromuscular disorders</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff11">
        <label>11</label>
        <institution>Department of General Medicine</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff12">
        <label>12</label>
        <institution>Hunan Engineering Research Center for Obesity and its Metabolic Complications</institution>
        <institution>Xiangya Hospital Central South University</institution>
        <addr-line>Changsha, Hunan</addr-line>
        <country>China</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Jing Wu <email>wujing0731@csu.edu.cn</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>9</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>12</volume>
      <elocation-id>e92486</elocation-id>
      <history>
        <date date-type="received">
          <day>2</day>
          <month>2</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>29</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>10</day>
          <month>8</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>11</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Yuchen He, Chen Chen, Rong Yin, Wuyang Zhang, Haoli Chang, Wei Yang, Fei Li, Xinhua Li, Zhuying Xia, Xiaoyun Xie, Jing Huang, Qiuming Zeng, Guang Yang, Jing Wu. Originally published in JMIR Medical Education (https://mededu.jmir.org), 09.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on https://mededu.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://mededu.jmir.org/2026/1/e92486" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Medical history taking (MHT) is a foundational clinical competency for medical students; however, traditional training models using standardized patients face challenges such as resource constraints. Large language model–powered virtual standardized patients (LLM-VSPs) offer a safe, repeatable platform for self-directed practice with AI-automated feedback. Nevertheless, their effectiveness in authentic teaching environments and underlying learning mechanisms require further investigation.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aims to evaluate the impact of an LLM-VSP system as an extracurricular self-practice tool on undergraduate medical students’ MHT performance in an authentic educational setting without disrupting routine instruction, and to further explore potential associations among practice behaviors, baseline proficiency, and intervention effects.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>This prospective cohort study enrolled 168 third-year medical students. Based on voluntary participation, students were assigned to an intervention group (n=120, using LLM-VSP) or a control group (n=48, receiving routine instruction). Propensity score matching (PSM) balanced confounding factors, yielding 40 matched pairs. Baseline MHT performance was assessed via virtual patient examination after didactic instruction but before clinical practicum. The primary outcome was end-of-term MHT performance assessed at an Objective Structured Clinical Examination station with real standardized patients. The differences between groups were compared using independent samples <italic>t</italic> test, with robustness validated through multiple linear regression, sensitivity analyses, and Rosenbaum bounds analyses. Exploratory analyses investigated the association between practice behaviors and scores, and observed benefit differences across baseline levels.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>After PSM, baseline characteristics were balanced (standardized mean difference &#60;0.1). The intervention group exhibited higher total MHT scores than controls (mean 87.71, SD 7.29 vs mean 83.74, SD 8.06; mean difference 3.98, 95% CI 0.55-7.40 points; <italic>P</italic>=.02; with a medium effect size of Cohen <italic>d</italic>=0.52). Advantages were observed in content (<italic>P</italic>=.03) and communication skills (<italic>P</italic>=.04) subscores. Regression analysis confirmed robust intervention effects (<italic>B</italic>=3.924, 95% CI 1.90-5.94; <italic>P</italic>&#60;.001; <italic>R</italic><sup>2</sup>=0.708), with sensitivity analyses supporting reliability. Exploratory analysis suggested that mere practice behavior metrics were not independent predictors of final scores, potentially being constrained by baseline proficiency, implying a “cognitive threshold” for effective AI-assisted training. Subgroup analysis indicated a trend of differential benefits: high baseline students appeared to show larger gains (matched sample: mean difference 5.93, 95% CI 2.17-9.70; overall sample: mean difference 7.04, 95% CI 3.84-10.25), whereas improvements in medium and low baseline students were relatively limited (matched sample: mean difference 2.94-2.99; overall sample: mean difference 1.29-1.64).</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>Introducing LLM-VSPs as a self-practice tool in diagnostics education may help improve undergraduate medical students’ MHT performance. Preliminary evidence suggests a potential “cognitive threshold,” implying students with solid theoretical foundations and higher baseline proficiency may better achieve skill transformation through AI-assisted autonomous practice. This provides a basis for future stratified teaching strategies and differentiated guidance for students of varying baseline levels.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>artificial intelligence</kwd>
        <kwd>education</kwd>
        <kwd>large language model</kwd>
        <kwd>medical history taking</kwd>
        <kwd>medical</kwd>
        <kwd>practice behaviors</kwd>
        <kwd>real-time feedback</kwd>
        <kwd>undergraduate</kwd>
        <kwd>virtual standardized patients</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Medical history taking (MHT) is an essential practical skill for clinicians in the diagnostic and treatment process [<xref ref-type="bibr" rid="ref1">1</xref>], and is also an important learning objective of the <italic>diagnostics</italic> course. MHT, as a core practical clinical competency, requires repeated deliberate practice for its mastery [<xref ref-type="bibr" rid="ref2">2</xref>]. Traditional MHT teaching predominantly relies on 2 primary pedagogical approaches: bedside teaching involving real patients and standardized patient (SP)–based teaching [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref5">5</xref>]. Yet the aforementioned instructional process remains plagued by an array of challenges and inherent limitations. Specifically, bedside teaching with real patients struggles to accommodate large-scale educational demands and carries the potential risk of exacerbating doctor-patient tensions; in contrast, SP-based teaching is confronted with issues such as substantial training costs, cumbersome management processes, difficulties in ensuring instructional standardization and homogeneity, and inadequacies in simulating authentic, complex clinical symptoms and signs [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>].</p>
      <p>Virtual standardized patients (VSPs) refer to virtual teaching tools constructed using digital technologies that can simulate the clinical characteristics of real patients and provide interactive feedback [<xref ref-type="bibr" rid="ref9">9</xref>], used for the training and assessment of medical students’ clinical diagnosis and treatment skills and communication abilities. VSPs can effectively address the limitations of SPs while exhibiting characteristics of repeatability and cost-effectiveness, providing new instructional modalities for medical education [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. However, early VSPs exhibited rigid interaction and were difficult to simulate the natural language communication of real patients [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Large language model–powered virtual standardized patients (LLM-VSPs) can deliver more naturalistic and authentic interactive experiences. They can provide medical students with highly realistic clinical consultation scenarios and timely, individualized feedback [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>], which endows them with great application potential in medical history collection, doctor-patient communication, and clinical reasoning training [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. However, the quality of LLM prompts directly affects the output performance of VSPs [<xref ref-type="bibr" rid="ref20">20</xref>]. Algorithmic biases and hallucinations in AI may lead to factually incorrect outputs [<xref ref-type="bibr" rid="ref21">21</xref>], which poses challenges to the reliability of automated AI scoring. Moreover, unsupervised autonomous practice with LLM-VSPs may foster communication behaviors that violate clinical ethics. Prior research on companion chatbots has documented not only AI-initiated inappropriate behaviors but also the potential for users to direct offensive or dismissive language toward AI [<xref ref-type="bibr" rid="ref22">22</xref>]. In educational contexts, such bidirectional misbehavior may erode rather than cultivate empathy, and risks reinforcing unprofessional communication habits.</p>
      <p>Current research on the effectiveness of LLM-VSPs in diagnostic courses remains relatively scarce, particularly studies that conduct in-depth analysis of LLM-VSPs’ usage behaviors and the corresponding learning outcomes among medical students, as well as exploration of differential benefits among students with different abilities. Accordingly, large-scale blind integration of this tool into teaching is premature. To clarify the real effectiveness and optimal application methods of LLM-VSPs, we conducted a small-scale pilot empirical exploration without interfering with the regular undergraduate teaching process. This study aims to conduct a prospective cohort study using voluntary recruitment combined with propensity score matching (PSM) to investigate 3 core research questions: first, whether incorporating LLM-VSPs into the auxiliary practice of MHT can improve medical students’ proficiency in this skill; second, what the correlation is between students’ practice behaviors during LLM-VSPs interactions and their subsequent learning outcomes; third, what the differential benefits are among student subgroups with different abilities from this intervention. The findings of this study are intended to provide empirical evidence and optimization directions for the future application of AI in MHT skill training.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Creation of LLM-VSPs</title>
        <p>This study used the DoctorU teaching system for MHT practice. This system constructed a multiagent workflow including LLM-VSPs agent, AI formative scoring agent, and AI feedback agent (<xref rid="figure1" ref-type="fig">Figure 1</xref>). The LLM-VSPs agent (see <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> for detailed prompts) was responsible for simulating patient roles. The AI formative scoring agent scored students’ single MHT performance based on a structured scoring rubric. After the MHT session, the system input the dialogue records, structured case facts (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), and corresponding structured scoring rubrics into the agent, which assessed whether students covered key MHT content through semantic judgment based on preset scoring points, score values, and scoring conditions, and generated scores, missed items, and brief evaluations. For synonymous questioning methods or nonstandard expressions, the system allowed semantic matching through a synonym library to determine whether students substantially addressed corresponding scoring points. The AI feedback agent generated immediate formative feedback after practice, combining dialogue records and scoring results, covering completed content, missed important information, deficiencies in MHT logic, and suggestions for subsequent improvement based on the feedback rule library. The system-maintained context within a single MHT session to ensure coherence of patient responses; however, individualized dialogue memories were not shared between different students and different training sessions to avoid cross-student information contamination. Simultaneously, the backend database logs the complete practice data of all students to support longitudinal tracking of individual learning growth and cross-sectional analysis of interaction performance and learning processes across students and cases. All system parameters remained fixed during the study period, and specific model invocation parameter settings are shown in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Architecture and workflow of the DoctorU large language model–powered virtual standardized patients (LLM-VSPs) system. ASR: automatic speech recognition; LLM: large language model; VSP: virtual standardized patient.</p>
          </caption>
          <graphic xlink:href="mededu_v12i1e92486_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>In accordance with the teaching syllabus and curriculum schedule of the diagnostic course for students in the 5-year clinical medicine program, and with reference to the textbook <italic>Diagnostics (10th Edition)</italic> published by the People’s Health Press, clinical cases were selected following a symptom-oriented approach. The criteria for case development were determined to primarily focus on newly admitted patients presenting with a single symptom, thereby ensuring uniformity in case difficulty levels. All cases did not involve real clinical patient data. Concurrently, a structured scoring rubric was developed, with a total score of 100 points. This rubric comprised 2 core dimensions: 80 points allocated to the completeness and accuracy of MHT content, and 20 points dedicated to the proficiency of MHT communication skills. Serving as a core functional module of the LLM-VSPs system, this rubric was designed to facilitate students’ completeness and logicality of MHT content through a cycle of “practice-assessment-feedback-correction.” Furthermore, given the existing constraints of AI systems in simulating nonverbal communication, fine-grained assessment of MHT interpersonal skills remained unfeasible [<xref ref-type="bibr" rid="ref23">23</xref>]; therefore, the scoring weight leaned more heavily toward content items. A total of 13 cases were developed: 11 for student practice and 2 for baseline assessment (see <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> for detailed cases and the scoring rubrics). Overall, 8 diagnostic course instructors of the current semester were invited to conduct a comprehensive review and refinement of all LLM-VSP cases, to safeguard the accuracy and instructional applicability of the cases.</p>
      </sec>
      <sec>
        <title>Participants</title>
        <p>The participants of this study consisted of 276 third-year medical students enrolled in the <italic>diagnostics</italic> course at Xiangya Hospital of Central South University during the spring semester of the 2024-2025 academic year. These students were from 3 cohorts: the 2022 entry class of the 5-year clinical medicine program, the 2022 entry class of the 5-year stomatology program, and the 2022 entry class of the 5-year anesthesiology program. All participating students had no previous training experience in MHT skills prior to the study. The inclusion criteria for this study were defined as follows: (1) being enrolled in the <italic>diagnostics</italic> course during the current semester and (2) voluntarily participating in this study and providing informed consent. The exclusion criteria were established as follows: (1) failure to complete the baseline assessment of MHT skills at the start of the semester and (2) failure to complete the postintervention assessment of MHT skills at the end of the semester. Upon completion of the research, 108 students were excluded (107 students failed to complete the baseline test and 1 student failed to complete the final test), and finally 168 students were included in data analysis.</p>
        <p>To determine student participation in the LLM-VSP practice sessions, a registration link was distributed to all students enrolled in this study. According to the principle of voluntary participation, all eligible students were assigned to one of the two groups: the experimental group (group A) received MHT training via LLM-VSPs–based practice, while the control group (group B) undertook conventional teaching for the same clinical skill training. However, according to the minimum effective practice criteria established in this study, among the students who had voluntarily registered for the LLM-VSPs–based practice group, 9 individuals failed to meet the aforementioned minimum practice threshold based on their practice records. They were deemed to have not effectively participated in the LLM-VSPs–based practice. According to the per-protocol analysis principle, the aforementioned students were reclassified into the control group. The final actual group allocation resulted in 120 students in the experimental group and 48 students in the control group (see <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> for the detailed participant screening flowchart).</p>
      </sec>
      <sec>
        <title>Intervention Measures</title>
        <p>In this study, students in group A used LLM-VSPs to practice history taking. In alignment with the curriculum schedule and teaching progress of the <italic>diagnostics</italic> course, starting from the third week of the semester, each student in group A was assigned 1 LLM-VSP–based clinical case per week, resulting in a total of 11 cases distributed throughout the intervention period. They could complete the autonomous practice of LLM-VSPs for history taking during their spare time through a mobile app. There were no restrictions on the number of practices and the duration of each practice. After each practice, the system would immediately generate AI-based assessment feedback. Students in group B did not use LLM-VSPs for the practice. Both groups of students received conventional teaching on MHT skills, including sessions with SPs and bedside practical training, in accordance with the predefined curriculum schedule of the <italic>diagnostics</italic> course.</p>
        <p>To ensure the authenticity and effectiveness of the LLM-VSPs intervention, we invited diagnostic course instructors of the current semester to conduct simulation tests. The results indicated that skillfully completing a single MHT session comprising basic self-introduction, chief complaint, history of present illness, past medical history, personal history, and family history required a minimum duration of 4 minutes. Furthermore, according to the LLM-VSPs structured scoring rubric, inquiring once for each module of the MHT content, including self-introduction, patient general information, chief complaint, history of present illness, and past medical history, would achieve a total score of 25 points. The research team unanimously agreed that practice sessions failing to meet the aforementioned criteria could not achieve the training objectives of structured MHT logic and content completeness and were considered ineffective practice. Therefore, this study established the minimum effective practice criteria as a single practice session duration of ≥4 minutes and an AI score of ≥25 points.</p>
      </sec>
      <sec>
        <title>Data Collection</title>
        <sec>
          <title>Baseline Characteristics</title>
          <p>Demographic data, including sex and academic major, were retrieved from the university’s student administration records. Sex was recorded as male or female. After completing the theoretical course of MHT and the practical training course of <italic>Medical Interview and Communication Skills</italic> at the start of the semester, 2 case-based MHT tests were administered to all participants via the DoctorU instructional system. The baseline score for each student was determined using the average of their AI-generated assessment scores from these 2 tests.</p>
        </sec>
        <sec>
          <title>Outcome Measures</title>
          <p>At the end of the semester, a postintervention assessment of MHT skills was administered to all participating students. The assessment was carried out in the format of Objective Structured Clinical Examination (OSCE) stations, with real SPs serving as simulated patients for the history-taking interaction. The assessment comprised 4 stations, each corresponding to a distinct clinical case scenario: coronary heart disease, pneumonia, urinary tract infection, and hyperthyroidism. Notably, none of these scenarios had been included in the LLM-VSP practice cases library prior to the assessment. All participants were randomly assigned by computer to one of the 4 assessment stations to complete the MHT assessment. Two standardized and trained examiners independently scored each candidate’s performance. The final assessment score was the average of the scores assigned by the 2 examiners. The intraclass correlation coefficient (ICC) based on a 2-way random effects model (absolute agreement) was used to evaluate interrater reliability, which was calculated independently within each assessment case. The scoring rubric was provided by the Department of Diagnostic Medicine, with a total score of 100 (including 60 points for MHT content and 40 points for MHT skills). This rubric served as a summative assessment tool, aiming to comprehensively evaluate students’ comprehensive MHT abilities after systematic diagnostic course learning; therefore, the scoring weights for MHT content and skills were set relatively balanced (see <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> for the specific scoring rubrics).</p>
        </sec>
        <sec>
          <title>LLM-VSPs Practice Data of Group A</title>
          <p>All students’ MHT practice data were recorded in detail via the backend of the DoctorU system. Unqualified practice data were excluded in accordance with the predefined minimum threshold for effective practice established earlier in the study. Ultimately, the LLM-VSPs–based MHT practice data of all students in group A were compiled, encompassing the following key metrics: the total number of effective practice sessions, the total number of effective practice cases, the average duration per practice session, the total duration of practice, the highest AI-generated assessment score, and the average AI-generated assessment score.</p>
          <p>Statistics</p>
          <p>Statistical analyses were performed using R software (version 4.4.2; R Foundation for Statistical Computing). PSM was implemented to mitigate selection bias arising from voluntary group allocation. Matching variables included sex, academic major, baseline test scores, and final test cases, with 1:1 nearest neighbor matching conducted between groups A and B using a caliper width of 0.1 SDs of the logit of the propensity score. Postmatching balance was evaluated through standardized mean differences (SMDs), where SMD&#60;0.1 indicated adequate balance. Final MHT assessment score comparisons between matched groups used independent samples <italic>t</italic> test or <italic>t’</italic> test, and Cohen <italic>d</italic> effect size was calculated. Multivariate linear regression adjusted for covariates was used to confirm intervention effect robustness. To assess the potential impact of unmeasured confounding factors on the results, the Rosenbaum bounds (Γ) were calculated. To verify the robustness of the primary findings against protocol deviations, a sensitivity analysis was conducted following the intention-to-treat (ITT) principle. Specifically, the 9 students who failed to meet the practice threshold were retained in the intervention group according to their original allocation. PSM and subsequent comparative analyses (including independent samples <italic>t</italic> tests and multivariate linear regression) were repeated on this ITT sample to ensure the consistency of the conclusions.</p>
          <p>An exploratory analysis was conducted within group A, using Pearson correlation analysis and partial correlation analysis (adjusted for baseline scores) to investigate the associations between practice behavior metrics and MHT scores, and a multivariate linear regression model was further constructed to explore the independent predictive effect of practice behaviors. The model included baseline scores as a covariate, and practice behavior metrics were selected based on the results of the preliminary correlation analyses.</p>
          <p>To investigate the heterogeneity of intervention effects across baseline proficiency levels, a generalized linear model was constructed. This model incorporated an interaction term between group assignment and baseline scores (as a continuous variable), while adjusting for sex, academic major, and final test case. This analysis was conducted in both the matched and overall samples. Furthermore, students were divided into high-, medium-, and low-level subgroups based on the 33rd and 67th percentiles of baseline scores. An exploratory subgroup analysis was performed to describe the outcome distributions across these subgroups. The subgroup models were adjusted for final test cases (with the overall sample additionally adjusting for baseline scores), excluding sex and academic major. Results are presented as adjusted means with SEs.</p>
          <p>All tests were 2-tailed with the significance threshold set at α=.05.</p>
        </sec>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>The study was reviewed by the Clinical Medical Ethics Review Committee of Xiangya Hospital of Central South University. The ethics committee reviewed the study in its entirety and concluded that the research was eligible for exemption from formal ethical review, as it complied with Specific Applicable Situations 1 and 2 of Article 32 of the Measures for Ethical Review of Life Science and Medical Research Involving Humans (2023 edition) issued by the National Health Commission of the People’s Republic of China [<xref ref-type="bibr" rid="ref24">24</xref>]. An official ethical review comment letter was issued (ethics committee record number 2026081836).</p>
        <p>Although the study was exempt from formal ethical review, comprehensive information regarding the research was fully provided, and written informed consent was secured from all participating students before enrollment.</p>
        <p>All data were deidentified prior to analysis. No individual student data or identifiable information are disclosed in this manuscript.</p>
        <p>Participants received no compensation.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Baseline Characteristics of the Participants</title>
        <p><xref ref-type="table" rid="table1">Table 1</xref> shows significant prematching differences between group A (n=120) and group B (n=48) in academic major distribution (<italic>P</italic>=.02) and baseline assessment scores (<italic>P</italic>=.01). Owing to computerized random station allocation, the distribution of final test cases was well-balanced between the two groups (<italic>P</italic>=.66). Following matching, 80 participants (40 in each group) were retained for final analysis. All SMDs between matched groups fell below 0.1, demonstrating balanced demographic characteristics (sex), academic majors, baseline test scores, and final test cases between the 2 groups (<xref rid="figure2" ref-type="fig">Figure 2</xref>).</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Baseline characteristics of participants before and after propensity score matching.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="110"/>
            <col width="90"/>
            <col width="90"/>
            <col width="90"/>
            <col width="0"/>
            <col width="80"/>
            <col width="0"/>
            <col width="70"/>
            <col width="0"/>
            <col width="70"/>
            <col width="0"/>
            <col width="80"/>
            <col width="0"/>
            <col width="80"/>
            <col width="0"/>
            <col width="80"/>
            <col width="0"/>
            <col width="70"/>
            <col width="0"/>
            <col width="0"/>
            <col width="60"/>
            <thead>
              <tr valign="top">
                <td colspan="2">Variables</td>
                <td colspan="8">Before matching</td>
                <td colspan="11">After matching</td>
                <td>SMD<sup>a</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">
                  <break/>
                </td>
                <td>Total (N=168)</td>
                <td>Group A (n=120)</td>
                <td>Group B (n=48)</td>
                <td colspan="2">Test statistic</td>
                <td colspan="2"><italic>P</italic> value</td>
                <td colspan="2">Total (n=80)</td>
                <td colspan="2">Group A (n=40)</td>
                <td colspan="2">Group B (n=40)</td>
                <td colspan="2">Test statistic</td>
                <td colspan="2"><italic>P</italic> value</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="6">
                  <bold>Sex</bold>
                  <bold>, n</bold>
                  <bold>(</bold>
                  <bold>%)</bold>
                </td>
                <td colspan="2">0.54 (1)<sup>b</sup></td>
                <td colspan="2">.46</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">0.06 (1)<sup>b</sup></td>
                <td colspan="2">.81</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Male</td>
                <td>61 (36.3)</td>
                <td>41 (34.2)</td>
                <td>20 (41.7)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">26 (32.5)</td>
                <td colspan="2">14 (35.0)</td>
                <td colspan="2">12 (30.0)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">0.050</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Female</td>
                <td>107 (63.7)</td>
                <td>79 (65.8)</td>
                <td>28 (58.3)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">54 (67.5)</td>
                <td colspan="2">26 (65.0)</td>
                <td colspan="2">28 (70.0)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">–0.050</td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>Academic</bold>
                  <bold>major</bold>
                  <bold>,</bold>
                  <bold>n</bold>
                  <bold>(</bold>
                  <bold>%)</bold>
                </td>
                <td colspan="2">7.74 (2)<sup>b</sup></td>
                <td colspan="2">.02</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">—<sup>c</sup></td>
                <td colspan="2">1.00</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Stomatology</td>
                <td>39 (23.2)</td>
                <td>32 (26.7)</td>
                <td>7 (14.6)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">16 (20.0)</td>
                <td colspan="2">9 (22.5)</td>
                <td colspan="2">7 (17.5)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">0.050</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Clinical medicine</td>
                <td>110 (65.5)</td>
                <td>71 (59.2)</td>
                <td>39 (81.3)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">59 (73.8)</td>
                <td colspan="2">28 (70.0)</td>
                <td colspan="2">31 (77.5)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">–0.075</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Anesthesiology</td>
                <td>19 (11.3)</td>
                <td>17 (14.2)</td>
                <td>2 (4.2)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">5 (6.3)</td>
                <td colspan="2">3 (7.5)</td>
                <td colspan="2">2 (5.0)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">0.025</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Baseline test score, mean (SD)</td>
                <td>63.2 (11.2)</td>
                <td>64.6 (10.8)</td>
                <td>59.7 (11.5)</td>
                <td colspan="2">2.57 (166)<sup>d</sup></td>
                <td colspan="2">.01</td>
                <td colspan="2">61.4 (10.3)</td>
                <td colspan="2">61.3 (9.2)</td>
                <td colspan="2">61.4 (11.4)</td>
                <td colspan="2">–0.04 (78)<sup>d</sup></td>
                <td colspan="2">.97</td>
                <td colspan="3">0.008</td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>Final</bold>
                  <bold>test case</bold>
                  <bold>, n</bold>
                  <bold>(</bold>
                  <bold>%)</bold>
                </td>
                <td colspan="2">1.59 (3)<sup>b</sup></td>
                <td colspan="2">.66</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">0.20 (3)<sup>b</sup></td>
                <td colspan="2">.98</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Pneumonia</td>
                <td>48 (28.6)</td>
                <td>34 (28.3)</td>
                <td>14 (29.2)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">23 (28.8)</td>
                <td colspan="2">12 (30.0)</td>
                <td colspan="2">11 (27.5)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">0.025</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>CHD<sup>e</sup></td>
                <td>42 (25.0)</td>
                <td>32 (26.7)</td>
                <td>10 (20.8)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">19 (23.8)</td>
                <td colspan="2">9 (22.5)</td>
                <td colspan="2">10 (25.0)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">–0.025</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Hyperthyroidism</td>
                <td>36 (21.4)</td>
                <td>23 (19.2)</td>
                <td>13 (27.1)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">19 (23.8)</td>
                <td colspan="2">10 (25.0)</td>
                <td colspan="2">9 (22.5)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">0.025</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>UTI<sup>f</sup></td>
                <td>42 (25.0)</td>
                <td>31 (25.8)</td>
                <td>11 (22.9)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">19 (23.8)</td>
                <td colspan="2">9 (22.5)</td>
                <td colspan="2">10 (25.0)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="3">–0.025</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>SMD: standardized mean difference. SMD value of less than 0.1 is widely recognized in clinical research, educational evaluation, and social science studies as a critical threshold for excellent baseline balance between 2 comparison groups.</p>
            </fn>
            <fn id="table1fn2">
              <p><sup>b</sup>Chi-square test (<italic>df</italic>).</p>
            </fn>
            <fn id="table1fn3">
              <p><sup>c</sup>For “academic major” after matching, the <italic>P</italic> value was calculated using the Fisher exact test, which does not produce a test statistic (indicated by “—”).</p>
            </fn>
            <fn id="table1fn4">
              <p><sup>d</sup><italic>t</italic> test (<italic>df</italic>).</p>
            </fn>
            <fn id="table1fn5">
              <p><sup>e</sup>CHD: coronary heart disease.</p>
            </fn>
            <fn id="table1fn6">
              <p><sup>f</sup>UTI: urinary tract infection.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Propensity score matching Love plot.</p>
          </caption>
          <graphic xlink:href="mededu_v12i1e92486_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Comparison of the Final MHT Assessment Scores</title>
        <p>Interrater reliability analysis of the final MHT assessment scores revealed ICCs across all 4 station cases ranging from 0.65 to 0.80 (Table S1 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>), indicating moderate to good agreement between examiners across the different stations. The postmatching comparison results of the final test scores for MHT between the 2 groups after intervention are presented in <xref ref-type="table" rid="table2">Table 2</xref>. The average total score of group A in MHT was 3.98 points higher than that of group B (mean 87.71, SD 7.29 vs mean 83.74, SD 8.06; <italic>P</italic>=.02; mean difference 3.98, 95% CI 0.55-7.40; Cohen <italic>d</italic>=0.52). Group A also demonstrated statistically significant advantages over group B in both the MHT content subscore (<italic>P</italic>=.03) and the MHT skill subscore of the assessment (<italic>P</italic>=.04).</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Comparison of the final medical history-taking (MHT) test scores between the 2 groups after propensity score matching.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="210"/>
            <col width="130"/>
            <col width="120"/>
            <col width="120"/>
            <col width="140"/>
            <col width="90"/>
            <col width="70"/>
            <col width="120"/>
            <thead>
              <tr valign="top">
                <td>Variables</td>
                <td>Total (n=80), mean (SD)</td>
                <td>Group A (n=40), mean (SD)</td>
                <td>Group B (n=40), mean (SD)</td>
                <td>Mean Difference (95% CI)</td>
                <td><italic>t</italic> test (<italic>df</italic>)</td>
                <td><italic>P</italic> value</td>
                <td>Cohen <italic>d</italic> (95% CI)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Total score of MHT (100 points)</td>
                <td>85.72 (7.90)</td>
                <td>87.71 (7.29)</td>
                <td>83.74 (8.06)</td>
                <td>3.98 (0.55-7.40)</td>
                <td>2.31 (78)</td>
                <td>.02</td>
                <td>0.52 (0.06-0.97)</td>
              </tr>
              <tr valign="top">
                <td>Score of MHT content (60 points)</td>
                <td>48.80 (5.93)</td>
                <td>50.23 (5.65)</td>
                <td>47.38 (5.93)</td>
                <td>2.85 (0.27-5.43)</td>
                <td>2.20 (78)</td>
                <td>.03</td>
                <td>0.49 (0.04-0.94)</td>
              </tr>
              <tr valign="top">
                <td>Score of MHT skills (40 points)</td>
                <td>36.92 (2.49)</td>
                <td>37.49 (2.13)</td>
                <td>36.36 (2.70)</td>
                <td>1.13 (0.04-2.21)</td>
                <td>2.07 (74.0)</td>
                <td>.04</td>
                <td>0.46 (0.01-0.91)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Robustness Test</title>
        <p>Multivariate linear regression confirmed the robustness of the above results (<xref ref-type="table" rid="table3">Table 3</xref>). In the unadjusted model, group A demonstrated significantly higher total MHT scores than group B (<italic>B</italic>=3.975; <italic>P</italic>=.02). Following adjustment for covariates (baseline test scores, sex, academic major, and final test cases), the intervention effect remained statistically significant (<italic>B</italic>=3.924; <italic>P</italic>&#60;.001). The model’s coefficient of determination <italic>R</italic><sup>2</sup> reached 0.708, indicating that the model explained the majority of the variance in final MHT assessment scores. In addition, to assess the potential threat of unmeasured confounding factors to the matching results, this study calculated the Rosenbaum bounds Γ=1.21 for this primary analysis sample.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Multivariate linear regression analysis of the final medical history-taking test scores (n=80).</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="30"/>
            <col width="170"/>
            <col width="210"/>
            <col width="110"/>
            <col width="100"/>
            <col width="130"/>
            <col width="110"/>
            <col width="0"/>
            <col width="110"/>
            <thead>
              <tr valign="top">
                <td colspan="3">Model and variables</td>
                <td><italic>B</italic><sup>a</sup> (95% CI)</td>
                <td>SE</td>
                <td>
                  <italic>β</italic>
                  <sup>b</sup>
                </td>
                <td><italic>t</italic> test (<italic>df</italic>)</td>
                <td><italic>P</italic> value</td>
                <td colspan="2">VIF<sup>c</sup> value</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="10">
                  <bold>Model 1 (without adjusting for covariates)<sup>d</sup></bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="2">Constant</td>
                <td>83.738 (81.318 to 86.157)</td>
                <td>1.215</td>
                <td>N/A<sup>e</sup></td>
                <td>68.898 (78)</td>
                <td>&#60;.001</td>
                <td colspan="2">N/A</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="2">Group A (reference group=group B)</td>
                <td>3.975 (0.553 to 7.397)</td>
                <td>1.719</td>
                <td>0.253</td>
                <td>2.313 (78)</td>
                <td>.02</td>
                <td colspan="2">N/A</td>
              </tr>
              <tr valign="top">
                <td colspan="10">
                  <bold>Model 2 (adjusted for covariates)<sup>f</sup></bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="2">Constant</td>
                <td>84.919 (77.770-92.068)</td>
                <td>3.585</td>
                <td>N/A</td>
                <td>23.686 (71)</td>
                <td>&#60;.001</td>
                <td colspan="2">N/A</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="2">Group A (reference group=group B)</td>
                <td>3.924 (1.904 to 5.944)</td>
                <td>1.013</td>
                <td>0.250</td>
                <td>3.873 (71)</td>
                <td>&#60;.001</td>
                <td colspan="2">1.007</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="2">Baseline test scores (points)</td>
                <td>0.012 (–0.089 to 0.114)</td>
                <td>0.051</td>
                <td>0.016</td>
                <td>0.245 (71)</td>
                <td>.81</td>
                <td colspan="2">1.039</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="2">Female (reference group=male)</td>
                <td>3.481 (1.200 to 5.761)</td>
                <td>1.144</td>
                <td>0.208</td>
                <td>3.043 (71)</td>
                <td>.003</td>
                <td colspan="2">1.065</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="8">
                  <bold>Academic</bold>
                  <bold>major</bold>
                  <bold>(</bold>
                  <bold>reference</bold>
                  <bold>g</bold>
                  <bold>roup=</bold>
                  <bold>stomatology</bold>
                  <bold>)</bold>
                </td>
                <td>1.040</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Clinical medicine</td>
                <td>2.911 (0.233 to 5.589)</td>
                <td>1.343</td>
                <td>0.163</td>
                <td>2.168 (71)</td>
                <td>.03</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Anesthesiology</td>
                <td>–0.528 (–5.196 to 4.140)</td>
                <td>2.341</td>
                <td>–0.016</td>
                <td>–0.226 (71)</td>
                <td>.82</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="8">
                  <bold>Final</bold>
                  <bold>test case (reference group</bold>
                  <bold>=</bold>
                  <bold>pneumonia</bold>
                  <bold>)</bold>
                </td>
                <td>1.041</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>CHD<sup>g</sup></td>
                <td>–15.854 (–18.765 to –12.942)</td>
                <td>1.460</td>
                <td>–0.860</td>
                <td>–10.856 (71)</td>
                <td>&#60;.001</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Hyperthyroidism</td>
                <td>–4.289 (–7.174 to –1.405)</td>
                <td>1.447</td>
                <td>–0.233</td>
                <td>–2.965 (71)</td>
                <td>.004</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>UTI<sup>h</sup></td>
                <td>–6.745 (–9.552 to –3.938)</td>
                <td>1.408</td>
                <td>–0.366</td>
                <td>–4.792 (71)</td>
                <td>&#60;.001</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>Unstandardized coefficient.</p>
            </fn>
            <fn id="table3fn2">
              <p><sup>b</sup>Standardized coefficient.</p>
            </fn>
            <fn id="table3fn3">
              <p><sup>c</sup>VIF: variance inflation factor.</p>
            </fn>
            <fn id="table3fn4">
              <p><sup>d</sup>Model parameters: <italic>R</italic><sup>2</sup>=0.064; Radj<sup>2</sup>=0.052; <italic>F</italic><sub>1, 78</sub>=5.348; <italic>P</italic>=.02.</p>
            </fn>
            <fn id="table3fn5">
              <p><sup>e</sup>N/A: not applicable.</p>
            </fn>
            <fn id="table3fn6">
              <p><sup>f</sup>Model parameters: <italic>R</italic><sup>2</sup>=0.708; Radj<sup>2</sup>=0.675; <italic>F</italic><sub>8,71</sub>=21.552; <italic>P</italic>&#60;.001.</p>
            </fn>
            <fn id="table3fn7">
              <p><sup>g</sup>CHD: coronary heart disease.</p>
            </fn>
            <fn id="table3fn8">
              <p><sup>h</sup>UTI: urinary tract infection.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Sensitivity Analysis</title>
        <p>According to the ITT analysis principle, the 9 students who were reassigned to the control group for failing to meet the practice threshold were reincluded in the intervention group (group A with 129 students and group B with 39 students), and PSM was reperformed to obtain 33 matched pairs. The analysis results are presented in Table S2 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>, revealing that group A’s MHT total scores remained significantly higher than those of group B (mean difference 5.27, 95% CI 1.52-9.02; <italic>P</italic>=.01; Cohen <italic>d</italic>=0.70). The multivariate linear regression results after adjusting for covariates remained significant (<italic>B</italic>=5.190, 95% CI 2.505-7.875; <italic>P</italic>&#60;.001). In addition, the Rosenbaum bounds Γ value for this sensitivity sample increased from 1.21 in the primary analysis to 1.42.</p>
      </sec>
      <sec>
        <title>Exploratory Analysis of LLM-VSPs Practice Data and the Final MHT Test Scores.</title>
        <p>Table S3 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref> summarizes the valid LLM-VSPs practice data for students in group A. On average, participants completed 6.25 valid practice sessions covering 5.21 unique clinical cases. Preliminary Pearson correlation analyses revealed that without adjustment for baseline scores, only total duration of practice (<italic>r</italic>=0.183; <italic>P</italic>=.046) and the average AI-generated assessment score (<italic>r</italic>=0.233; <italic>P</italic>=.01) exhibited weak positive linear associations with final MHT test scores. All other practice metrics failed to demonstrate statistically significant linear correlations with the outcome. To exclude baseline scores as an important confounding factor, we further conducted partial correlation analysis (Table S3 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). The results showed that after controlling for baseline scores, the correlations of both the total duration of practice (<italic>r<sub>partial</sub></italic>=0.075; <italic>P</italic>=.42) and the average AI-generated assessment score (<italic>r<sub>partial</sub></italic>=0.097; <italic>P</italic>=.29) with final MHT test scores disappeared. Further multiple linear regression analysis confirmed (<xref ref-type="table" rid="table4">Table 4</xref>) that after adjusting for baseline, neither the total duration of practice (<italic>β</italic>=.058; <italic>P</italic>=.56) nor the average AI-generated assessment score (<italic>β</italic>=.096; <italic>P</italic>=.38) could independently predict final test scores, whereas baseline scores showed a marginal predictive trend (<italic>β</italic>=.209; <italic>P</italic>=.06). Although the overall regression model showed statistical significance (<italic>F</italic><sub>3, 116</sub>=3.947; <italic>P</italic>=.01), the adjusted coefficient of determination Radj<sup>2</sup>=0.069 was at a low level.</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Multiple linear regression analysis predicting the final medical history-taking (MHT) test scores in group A (n=120).<sup>a</sup></p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="340"/>
            <col width="220"/>
            <col width="100"/>
            <col width="110"/>
            <col width="140"/>
            <col width="90"/>
            <thead>
              <tr valign="top">
                <td>Variables</td>
                <td><italic>B</italic> (95% CI)</td>
                <td>SE</td>
                <td>
                  <italic>β</italic>
                </td>
                <td><italic>t</italic> test (<italic>df</italic>)</td>
                <td><italic>P</italic> value</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Constant</td>
                <td>71.011 (60.515 to 81.508)</td>
                <td>5.300</td>
                <td>N/A<sup>b</sup></td>
                <td>13.400 (116)</td>
                <td>&#60;.001</td>
              </tr>
              <tr valign="top">
                <td>Baseline test scores (points)</td>
                <td>0.146 (–0.004 to 0.296)</td>
                <td>0.076</td>
                <td>0.209</td>
                <td>1.922 (116)</td>
                <td>.06</td>
              </tr>
              <tr valign="top">
                <td>Total duration of LLM-VSPs<sup>c</sup>–based MHT practice sessions (minutes)<sup>d</sup></td>
                <td>0.006 (–0.015 to 0.027)</td>
                <td>0.010</td>
                <td>0.058</td>
                <td>0.582 (116)</td>
                <td>.56</td>
              </tr>
              <tr valign="top">
                <td>Average AI-generated assessment score for LLM-VSPs–based MHT<sup>d</sup></td>
                <td>0.076 (–0.093 to 0.246)</td>
                <td>0.086</td>
                <td>0.096</td>
                <td>0.891 (116)</td>
                <td>.38</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table4fn1">
              <p><sup>a</sup>Model parameters: <italic>R</italic><sup>2</sup>=0.093; Radj<sup>2</sup>=0.069; <italic>F</italic><sub>3, 116</sub>=3.947; <italic>P</italic>=.01.</p>
            </fn>
            <fn id="table4fn2">
              <p><sup>b</sup>N/A: not applicable.</p>
            </fn>
            <fn id="table4fn3">
              <p><sup>c</sup>LLM-VSP: large language model–powered virtual standardized patient.</p>
            </fn>
            <fn id="table4fn4">
              <p><sup>d</sup>The practice behavior metrics were selected as predictors based on their significant associations observed in the preliminary unadjusted correlation analyses.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Baseline Heterogeneity Tests and Subgroup Analyses</title>
        <p>Results from the interaction term analysis regarding baseline heterogeneity suggested that the interaction between group assignment and baseline status did not reach statistical significance in either the matched or overall samples (matched sample: <italic>P</italic>=.63; overall sample: <italic>P</italic>=.20; see Tables S4 and S5 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). However, the main effect of group assignment indicated that students in the LLM-VSP group had significantly higher final exam scores than those in the control group in both the matched sample (mean difference 3.92, SE 1.02, 95% CI 1.89-5.95; <italic>P</italic>&#60;.001) and the overall sample (mean difference 3.61, SE 0.99, 95% CI 1.65-5.57; <italic>P</italic>&#60;.001). Exploratory subgroup analyses (<xref ref-type="table" rid="table5">Table 5</xref>) indicated that the between-group mean difference in the high baseline subgroup appeared to be 5.93 (95% CI 2.17-9.70) in the matched sample and 7.04 (95% CI 3.84-10.25) in the overall sample. In contrast, the mean differences observed in the low and moderate baseline subgroups appeared relatively smaller (matched sample: 2.94-2.99; overall sample: 1.29-1.64).</p>
        <table-wrap position="float" id="table5">
          <label>Table 5</label>
          <caption>
            <p>Exploratory descriptive analysis of the final medical history-taking test scores by baseline tertiles in overall unmatched and matched samples.<sup>a</sup></p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="30"/>
            <col width="220"/>
            <col width="0"/>
            <col width="160"/>
            <col width="0"/>
            <col width="310"/>
            <col width="0"/>
            <col width="0"/>
            <col width="250"/>
            <col width="0"/>
            <col width="0"/>
            <thead>
              <tr valign="top">
                <td colspan="4">Sample, subgroup, and group</td>
                <td colspan="2">Baseline, mean (SD)</td>
                <td colspan="2">Adjusted final, mean (SE; 95% CI)</td>
                <td colspan="3">Adjusted final (group A – group B), mean difference (95% CI)</td>
                <td>
                  <break/>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="12">
                  <bold>Matched sample</bold>
                  <bold>(</bold>
                  <bold>n=80)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="8">
                  <bold>Low tertile</bold>
                </td>
                <td colspan="3">2.99 (–1.80 to 7.77)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group A (n=12)</td>
                <td colspan="2">49.73 (5.10)</td>
                <td colspan="2">87.33 (1.70; 83.80-90.87)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group B (n=14)</td>
                <td colspan="2">N/A<sup>b</sup></td>
                <td colspan="2">84.35 (1.53; 81.16-87.53)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="8">
                  <bold>Medium tertile</bold>
                </td>
                <td colspan="3">2.94 (–0.70 to 6.58)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group A (n=14)</td>
                <td colspan="2">61.56 (3.15)</td>
                <td colspan="2">87.00 (1.20; 84.50-89.49)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group B (n=13)</td>
                <td colspan="2">N/A</td>
                <td colspan="2">84.05 (1.28; 81.41-86.70)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="8">
                  <bold>High tertile</bold>
                </td>
                <td colspan="3">5.93 (2.17 to 9.70)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group A (n=14)</td>
                <td colspan="2">72.42 (5.18)</td>
                <td colspan="2">88.14 (1.24; 85.56-90.72)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group B (n=13)</td>
                <td colspan="2">N/A</td>
                <td colspan="2">82.21 (1.26; 79.60-84.82)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Overall unmatched sample</bold>
                  <bold>(</bold>
                  <bold>N=168)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="8">
                  <bold>Low tertile</bold>
                </td>
                <td colspan="3">1.64 (–2.36 to 5.63)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group A (n=33)</td>
                <td colspan="2">50.81 (6.28)</td>
                <td colspan="2">84.27 (1.29; 81.69-86.86)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group B (n=23)</td>
                <td colspan="2">N/A</td>
                <td colspan="2">82.64 (1.51; 79.61-85.66)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="8">
                  <bold>Medium tertile</bold>
                </td>
                <td colspan="3">1.29 (–2.37 to 4.94)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group A (n=42)</td>
                <td colspan="2">63.56 (2.60)</td>
                <td colspan="2">85.79 (0.88; 84.02-87.56)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group B (n=14)</td>
                <td colspan="2">N/A</td>
                <td colspan="2">84.50 (1.57; 81.34-87.66)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="8">
                  <bold>High tertile</bold>
                </td>
                <td colspan="3">7.04 (3.84 to 10.25)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group A (n=45)</td>
                <td colspan="2">75.27 (5.54)</td>
                <td colspan="2">88.41 (0.70; 87.02-89.81)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Group B (n=11)</td>
                <td colspan="2">N/A</td>
                <td colspan="2">81.37 (1.44; 78.49-84.25)</td>
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table5fn1">
              <p><sup>a</sup>Matched sample models adjusted for final test case; overall unmatched sample models adjusted for final test case and baseline score. Given that the interaction analysis showed test case explained the largest proportion of variance (partial η<sup>2</sup>&#62;0.43), whereas sex and major contributed minimally, these variables were not included in the models.</p>
            </fn>
            <fn id="table5fn2">
              <p><sup>b</sup>N/A: not applicable.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Overview</title>
        <p>This study introduced LLM-VSPs as a self-directed practice tool for MHT in an authentic educational setting and evaluated their effectiveness through PSM and multidimensional robustness checks. The results indicated that the integration of LLM-VSPs effectively enhanced medical students’ MHT competency. Nevertheless, exploratory analysis suggests that this benefit may not be uniformly distributed across student cohorts with varying baseline proficiencies. Specifically, while students across all proficiency levels appeared to benefit from the practice, the high baseline students with a robust theoretical knowledge foundation tended to show a greater magnitude of improvement, whereas the gains for students with medium to low baseline levels seemed relatively limited. These observations imply that the effectiveness of LLM-driven clinical skills training tools may be influenced by learners’ prior knowledge levels, offering critical reference points for the practical deployment of such tools and the formulation of stratified teaching strategies.</p>
      </sec>
      <sec>
        <title>Intervention Efficacy and Robustness</title>
        <p>Analysis of the matched samples revealed that the introduction of LLM-VSPs had a significant positive effect on enhancing medical students’ MHT skills (mean difference 3.98; <italic>P</italic>=.02). The positive effect remained robust (<italic>B</italic>=3.924; <italic>P</italic>&#60;.001) even after further controlling for known confounders using a multiple linear regression model. For interventional research conducted in undergraduate teaching settings, randomized controlled trials (RCTs) represent the gold standard for causal inference. Nevertheless, random allocation of educational resources may undermine teaching homogeneity and educational equity. Therefore, this study adopted a prospective cohort design based on voluntary registration and PSM.</p>
        <p>PSM [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>] can simulate the randomization process of an RCT, reconstructing a new sample set with balanced baseline characteristics and thus establishing a methodological foundation for between-group comparisons. However, controlling for observed confounders is not equivalent to eliminating hidden bias [<xref ref-type="bibr" rid="ref27">27</xref>]. A Rosenbaum bounds analysis yielded a Γ value of merely 1.21, suggesting that the conclusions remain somewhat sensitive to relatively minor hidden biases. Given the voluntary enrollment design, learning motivation is the most prominent potential unobserved confounder [<xref ref-type="bibr" rid="ref28">28</xref>]. This raises the question of whether the outstanding performance of the intervention group was influenced by stronger learning motivation rather than the actual effect of the LLM-VSPs. Results from the ITT sensitivity analysis provide tentative supportive evidence. By retaining 9 students who voluntarily enrolled but failed to meet the effective practice standards—implying relatively lower motivation—in the intervention group, we artificially diluted the group’s motivational advantage. Instead of diminishing, the between-group difference further expanded (mean difference 3.98, 95% CI 0.55-7.40 vs mean difference 5.27, 95% CI 1.52-9.02), the effect size increased from 0.52 to 0.70, and the Rosenbaum bounds Γ value rose to 1.42. The results suggest that motivational bias likely did not inflate the intervention effect. However, we must interpret this finding cautiously. Since rematching altered the sample composition (from 40 to 33 pairs) and these 9 students might differ in attributes other than motivation, we cannot rely solely on this sensitivity analysis to definitively rule out hidden bias on the intervention effects. Nevertheless, these findings still provide supplementary support for the robustness of the LLM-VSPs intervention effects.</p>
        <p>This finding aligns with the results of Luo et al [<xref ref-type="bibr" rid="ref29">29</xref>] regarding virtual patients improving ophthalmology history-taking skills. LLM-VSPs provide medical students with a secure, repeatable, and accessible practice platform [<xref ref-type="bibr" rid="ref30">30</xref>], allowing for repeated trial-and-error and consolidation of learning, thereby helping them improve their MHT competency. Furthermore, real human SPs possess greater clinical contextual authenticity and induce performance pressure; students may experience anxiety that impedes knowledge retrieval and expression. Repeated simulated interactions with LLM-VSPs can also enhance students’ self-efficacy, enabling them to remain more composed when facing real SPs [<xref ref-type="bibr" rid="ref31">31</xref>].</p>
      </sec>
      <sec>
        <title>Dimensional Differences in Efficacy</title>
        <p>A detailed analysis of the dimensional characteristics of the LLM-VSPs intervention effects revealed that the MHT score improvements facilitated by LLM-VSPs were primarily reflected in the content dimension, while the enhancement in the skills dimension was relatively limited. This finding aligns closely with the discrepancy in scoring weights between the formative assessment [<xref ref-type="bibr" rid="ref32">32</xref>] via LLM-VSPs and the end-of-term summative assessment. Such a weighting discrepancy is not a contradiction in the study design but rather stems from the differing evaluation purposes and technical media of the two assessments. LLM-VSPs focus on helping beginners master the correct MHT content framework through a “practice-assessment-feedback-correction” cycle, with the scoring emphasizing the completeness and accuracy of the consultation content. Furthermore, constrained by the current limitations of LLM technology, LLM-VSPs struggle to simulate nonverbal communication elements during consultations, such as eye contact, gestures, and nodding. Students also exhibit challenges in conveying genuine affective reactions in their interactions with LLM-VSPs [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. Consequently, the scoring and feedback primarily concentrate on the quantitative assessment of content completeness. In contrast, the end-of-term summative evaluation based on human SPs aims to comprehensively assess students’ overall clinical competence, with more balanced weights assigned to both dimensions. This outcome clearly delineates the current applicability of LLM-VSPs as a training tool for MHT skills. While LLM-VSPs predominantly fulfill the training needs in the content dimension of MHT, they fall short in providing effective practice and precise feedback for interviewing techniques, including the fluency and accessibility of language, empathy, various nonverbal communications [<xref ref-type="bibr" rid="ref35">35</xref>], and responsiveness to patients’ emotions [<xref ref-type="bibr" rid="ref36">36</xref>]. This finding resonates with recent research [<xref ref-type="bibr" rid="ref23">23</xref>] conclusions regarding the application of conversational AI in medical education, which indicate that due to current technical bottlenecks in affective computing and multimodal interaction, AI systems are ill-equipped to handle the comprehensive training of communication skills. Nevertheless, with the rapid advancement of LLMs equipped with visual and auditory perception capabilities, future LLM-VSPs are expected to achieve breakthroughs in simulating nonverbal interactions [<xref ref-type="bibr" rid="ref37">37</xref>], thereby enabling their application in more comprehensive interviewing skills practice.</p>
      </sec>
      <sec>
        <title>Practice Behaviors and the Cognitive Threshold</title>
        <p>Although the preceding analysis indicated the intervention effects of LLM-VSPs on medical students, the learning mechanisms underlying whether merely providing LLM-VSPs practice tools can lead to skill improvement require further investigation. To this end, this study analyzed the association between LLM-VSP practice behaviors and final MHT scores at a deeper level. We observed a thought-provoking phenomenon: before adjusting for baseline scores, the average AI score was weakly positively correlated with the final exam score, while practice frequency and duration showed no significant association, seemingly corroborating the notion that “practice quality matters more than practice quantity.” However, after controlling for baseline scores, the significant correlation between the average AI score and the final MHT scores disappeared. Multiple linear regression further indicated that compared to LLM-VSP practice behavior indicators, baseline scores had a stronger predictive power for medical students’ final MHT competence. This suggests that in the self-directed practice mode, merely increasing practice duration or frequency may not necessarily lead to competence improvement, and AI scores appear to be strongly associated with students’ prior knowledge levels. Admittedly, these results involved multiple comparisons without strict statistical correction, merely identifying preliminary associations between practice behaviors, baseline scores, and final MHT competence.</p>
        <p>Based on the above findings, we speculate that the self-directed use of AI-empowered practice tools may involve a “cognitive threshold.” The acquisition of MHT skills is essentially the transformation from “declarative knowledge” to “procedural skills” [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. Baseline assessments mainly reflect the stock of declarative knowledge at the “remembering and understanding” cognitive tier shortly after theoretical coursework, while the final examination demands students deploy procedural skills at the “application and analysis” level in authentic interpersonal clinical encounters. The transition from “knowing what to ask” to “being able to ask fluently” not only necessitates sustained simulated practice [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>] but, more importantly, requires the prerequisite of mastering solid declarative knowledge.</p>
        <p>While LLM-VSPs deliver real-time quantitative feedback that precisely identifies omissions in history-taking content [<xref ref-type="bibr" rid="ref32">32</xref>] and facilitates deliberate practice [<xref ref-type="bibr" rid="ref41">41</xref>], learners lacking a robust foundational theoretical framework may fail to thoroughly comprehend the diagnostic reasoning underpinning AI-generated feedback. Instead, they may merely compile inquiry items in a passive manner, confining their training to superficial mechanical repetition. This hinders efficient knowledge transfer and internalization [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>]. In contrast, students with higher baseline proficiency have already mastered the theoretical principles of history-taking and possess the cognitive capacity to thoroughly interpret AI-generated feedback [<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. This allows them to leverage such feedback to rectify omissions in clinical inquiry and remedy knowledge deficits within a virtuous “practice-feedback-correction” cycle. This may be the prerequisite for AI-supported deliberate practice to take effect [<xref ref-type="bibr" rid="ref41">41</xref>]: effective practice requires not only repetition but also the correction of errors during the process.</p>
      </sec>
      <sec>
        <title>Differential Effects by Baseline Proficiency</title>
        <p>To verify the above speculation, we further analyzed the potential heterogeneous effect of baseline ability on the LLM-VSP intervention. Strict interaction term tests showed that the interaction between baseline scores and group assignment did not reach statistical significance in either the matched or full samples, suggesting that LLM-VSPs appeared to demonstrate general effectiveness across different baseline levels without presenting significant statistical differences. However, given the limited sample size of this study, which might affect the statistical power of the interaction term analysis [<xref ref-type="bibr" rid="ref46">46</xref>], we conducted an exploratory subgroup analysis to observe the data characteristics of final exam scores across different baseline levels in greater detail. Considering that PSM might lead to the loss of some high baseline samples [<xref ref-type="bibr" rid="ref25">25</xref>], we analyzed both the full sample and the matched sample. The results revealed that while students in all baseline subgroups appeared to benefit from the use of LLM-VSPs, the high baseline subgroup showed a trend of larger between-group differences, whereas the differences in the medium and low baseline subgroups were relatively smaller. Moreover, this trend appeared more pronounced in the full sample compared to the matched sample. It should be cautiously noted that this result might be influenced by the small sample size of the control group in that subgroup (n=11). Additionally, the ICC values for several stations in this study were relatively low; such measurement error might weaken the statistical power to some extent, making it difficult to detect potential subtle skill improvements among medium and low baseline students. Nevertheless, it still aligns with our speculation: students in the high baseline group seemed to benefit more from the self-directed practice with LLM-VSPs.</p>
        <p>Furthermore, if this trend of differential benefit holds true in larger samples, it offers a plausible explanation for why the overall effect size was only moderate. Given that students with medium or low baseline proficiency constitute the majority of the sample, their relatively limited benefit magnitudes may have diluted the larger benefits observed among the high baseline subgroup, thereby narrowing the overall mean difference. Admittedly, we must acknowledge that since the interaction term analysis did not reach statistical significance, the above heterogeneity analysis remains a trend observation, and future studies with larger samples and more rigorous designs are needed for confirmation.</p>
      </sec>
      <sec>
        <title>Implications for Stratified Teaching</title>
        <p>These findings offer important reference points for the application of LLM-VSPs in <italic>diagnostics</italic> education. As an after-class self-directed learning tool, LLM-VSPs can enhance medical students’ MHT competency. However, maximizing the pedagogical benefits of this tool requires further reflection in teaching practice. Based on the trends observed in the aforementioned analysis, we speculate that merely increasing practice duration or frequency may not necessarily guarantee competence improvement, and the effectiveness of practice quality might be constrained by baseline ability. Therefore, paying attention to students’ baseline levels and implementing stratified interventions may help optimize teaching outcomes. For students who already possess a solid theoretical foundation, self-directed use of LLM-VSPs can facilitate the transformation of knowledge into skills. Nevertheless, for learners with medium or low baseline proficiency, structured guided support may be required to replace independent exploration to maximize the intervention benefits delivered by LLM-VSPs. Specifically, using LLM-VSPs under teacher guidance could help these students consolidate their theoretical groundwork and interpret AI feedback, thereby achieving a transition from mechanical repetition to effective practice. However, further empirical research is required to verify whether this hybrid teaching model combining instructor guidance and LLM-VSPs practice can substantially improve MHT proficiency among students with medium and low baseline ability.</p>
        <p>Furthermore, the significance of LLM-VSPs extends beyond their immediate educational outcomes to their potential in promoting digital equity in medical education. Even in the current era of widespread AI accessibility, underresourced grassroots medical schools often struggle with limited faculty and SP resources. In these settings, educators face challenges in transforming general-purpose LLMs into effective pedagogical tools [<xref ref-type="bibr" rid="ref47">47</xref>] and lack the capacity to undertake the laborious training required for human SPs [<xref ref-type="bibr" rid="ref48">48</xref>]. The standardized LLM-VSPs agent developed in this study is accessible via a WeChat miniprogram. It not only eliminates development and training costs for grassroots medical institutions but also enables students at these institutions to receive high-quality history-taking training comparable to that offered by top-tier medical schools using only a smartphone and internet access. However, the accessibility of a tool does not equate to the efficiency of learning. How to assist learners of varying proficiency levels to derive equally effective practice benefits may be the key to unlocking the potential of AI teaching tools and realizing high-quality educational equity.</p>
      </sec>
      <sec>
        <title>Advantages and Limitations</title>
        <p>This study validated the effectiveness of LLM-VSPs within an authentic diagnostic teaching context. It adopted PSM and multiple linear regression analysis methods to minimize confounding bias to the greatest extent, providing a methodological paradigm for causal inference in educational settings where RCTs were impractical. Simultaneously, this study extended beyond simple verification of intervention efficacy; rather, it conducted an in-depth analysis combining various behavioral data from practice with baseline levels. It preliminarily revealed the “cognitive threshold” phenomenon during the self-directed use of LLM-VSPs and identified a trend of differential benefits across different groups, providing data for the subsequent implementation of refined teaching with LLM-VSPs.</p>
        <p>Admittedly, this study has certain limitations. First, although we used various statistical methods to control for confounding factors, Rosenbaum bounds sensitivity analysis indicated that our results remained comparatively vulnerable to unmeasured confounders; future studies require more rigorous experimental designs to draw definitive causal inferences. Second, this study was conducted as a single-center investigation with a limited sample size, and PSM also led to a reduction in sample size, which may affect the statistical power and generalizability of the research conclusions. Third, the analyses concerning the association between practice behaviors and final history-taking competency, as well as baseline heterogeneity, were exploratory in nature. The “cognitive threshold” hypothesis and the trend of “greater benefits for high baseline students” proposed in this study were primarily based on trend observations and lack robust statistical support evidence; thus, their causal mechanisms remain to be clarified. In future research, multicenter, large-sample RCTs should be conducted to generate more robust evidence in support of the findings. In addition, the LLM-VSPs agent developed herein has undergone iterative testing and refinement throughout its development phase to secure accurate automated scoring. Nevertheless, follow-up validation studies are still needed to formally establish its measurement reliability.</p>
      </sec>
      <sec>
        <title>Conclusion</title>
        <p>Based on PSM analysis, this study has indicated that introducing LLM-VSPs as an after-class self-directed practice tool in <italic>diagnostics</italic> education may help improve medical students’ MHT competency. Exploratory analysis further suggests a trend that the educational benefits of LLM-VSPs appear more pronounced among high baseline students with firm theoretical foundations, while the improvement for medium or low baseline students is relatively limited. This phenomenon implies that during the self-directed use of LLM-driven learning tools, learners’ prior knowledge levels may constrain the efficiency of using instant AI feedback to achieve the transformation from declarative knowledge into procedural skills. Accordingly, we suggest that future teaching practices should consider students’ differences in baseline competencies and explore teaching strategies for stratified interventions when deploying LLM-VSPs. High baseline students might attempt self-directed use, whereas for medium and low baseline students, it is necessary to explore a guided teaching model combining instructor guidance and LLM-VSPs. Future research using multicenter, large-sample RCTs is needed to further verify the aforementioned differences in benefits and the effectiveness of stratified teaching strategies.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Development of the DoctorU teaching system for medical history-taking practice.</p>
        <media xlink:href="mededu_v12i1e92486_app1.docx" xlink:title="DOCX File , 43 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Scoring rubrics for virtual standardized patient cases.</p>
        <media xlink:href="mededu_v12i1e92486_app2.docx" xlink:title="DOCX File , 230 KB"/>
      </supplementary-material>
      <supplementary-material id="app3">
        <label>Multimedia Appendix 3</label>
        <p>The detailed participant screening flowchart.</p>
        <media xlink:href="mededu_v12i1e92486_app3.docx" xlink:title="DOCX File , 57 KB"/>
      </supplementary-material>
      <supplementary-material id="app4">
        <label>Multimedia Appendix 4</label>
        <p>Scoring rubrics for the final medical history-taking test.</p>
        <media xlink:href="mededu_v12i1e92486_app4.docx" xlink:title="DOCX File , 32 KB"/>
      </supplementary-material>
      <supplementary-material id="app5">
        <label>Multimedia Appendix 5</label>
        <p>Supplementary statistical analyses.</p>
        <media xlink:href="mededu_v12i1e92486_app5.docx" xlink:title="DOCX File , 30 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">ICC</term>
          <def>
            <p>intraclass correlation coefficient</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">ITT</term>
          <def>
            <p>intention-to-treat</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">LLM-VSP</term>
          <def>
            <p>large language model–powered virtual standardized patient</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">MHT</term>
          <def>
            <p>medical history taking</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">OSCE</term>
          <def>
            <p>Objective Structured Clinical Examination</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">PSM</term>
          <def>
            <p>propensity score matching</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">RCT</term>
          <def>
            <p>randomized controlled trial</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">SMD</term>
          <def>
            <p>standardized mean difference</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">SP</term>
          <def>
            <p>standardized patient</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb11">VSP</term>
          <def>
            <p>virtual standardized patient</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors acknowledge the tireless educational efforts of diagnostics faculty in delivering core clinical competencies and participating students for their proactive engagement with this pedagogical innovation.</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>This work was supported by the Department of Education of Hunan Province (grant 202401000359) and Central South University (grants 2025jy024, 2025ALK023, 2024jy178, and 2024jy051-1).</p>
      </sec>
    </notes>
    <notes>
      <sec>
        <title>Data Availability</title>
        <p>The datasets generated or analyzed during this study are available from the corresponding author on reasonable request.</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Conceptualization: YH (lead), CC (equal)</p>
        <p>Data curation: YH (lead), RY (equal), HC (supporting)</p>
        <p>Formal analysis: YH (lead), RY (supporting)</p>
        <p>Funding acquisition: JW</p>
        <p>Investigation: YH (lead), CC (equal)</p>
        <p>Methodology: YH (lead), CC (equal)</p>
        <p>Project administration: JW (lead), YH (supporting), CC (supporting)</p>
        <p>Resources: JW</p>
        <p>Software: YH (lead), CC (equal), WY, FL, XL, ZX, XX, JH, QZ, GY (all supporting)</p>
        <p>Supervision: JW</p>
        <p>Validation: YH (lead), CC (equal)</p>
        <p>Visualization: YH (lead), CC (equal)</p>
        <p>Writing—original draft: YH</p>
        <p>Writing—review and editing: YH (lead), JW (equal), CC (supporting), WZ (supporting)</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Keifenheim</surname>
              <given-names>KE</given-names>
            </name>
            <name name-style="western">
              <surname>Teufel</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ip</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Speiser</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Leehr</surname>
              <given-names>EJ</given-names>
            </name>
            <name name-style="western">
              <surname>Zipfel</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Herrmann-Werner</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Teaching history taking to medical students: a systematic review</article-title>
          <source>BMC Med Educ</source>
          <year>2015</year>
          <volume>15</volume>
          <fpage>159</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-015-0443-x"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-015-0443-x</pub-id>
          <pub-id pub-id-type="medline">26415941</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-015-0443-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC4587833</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Elendu</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Amaechi</surname>
              <given-names>DC</given-names>
            </name>
            <name name-style="western">
              <surname>Okatta</surname>
              <given-names>AU</given-names>
            </name>
            <name name-style="western">
              <surname>Amaechi</surname>
              <given-names>EC</given-names>
            </name>
            <name name-style="western">
              <surname>Elendu</surname>
              <given-names>TC</given-names>
            </name>
            <name name-style="western">
              <surname>Ezeh</surname>
              <given-names>CP</given-names>
            </name>
            <name name-style="western">
              <surname>Elendu</surname>
              <given-names>ID</given-names>
            </name>
          </person-group>
          <article-title>The impact of simulation-based training in medical education: a review</article-title>
          <source>Medicine (Baltimore)</source>
          <year>2024</year>
          <volume>103</volume>
          <issue>27</issue>
          <fpage>e38813</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.ovid.com/10.1097/MD.0000000000038813"/>
          </comment>
          <pub-id pub-id-type="doi">10.1097/MD.0000000000038813</pub-id>
          <pub-id pub-id-type="medline">38968472</pub-id>
          <pub-id pub-id-type="pii">00005792-202407050-00022</pub-id>
          <pub-id pub-id-type="pmcid">PMC11224887</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>May</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>JP</given-names>
            </name>
          </person-group>
          <article-title>A ten-year review of the literature on the use of standardized patients in teaching and learning: 1996-2005</article-title>
          <source>Med Teach</source>
          <year>2009</year>
          <volume>31</volume>
          <issue>6</issue>
          <fpage>487</fpage>
          <lpage>492</lpage>
          <pub-id pub-id-type="doi">10.1080/01421590802530898</pub-id>
          <pub-id pub-id-type="medline">19811163</pub-id>
          <pub-id pub-id-type="pii">10.1080/01421590802530898</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Barrows</surname>
              <given-names>HS</given-names>
            </name>
          </person-group>
          <article-title>An overview of the uses of standardized patients for teaching and evaluating clinical skills. AAMC</article-title>
          <source>Acad Med</source>
          <year>1993</year>
          <volume>68</volume>
          <issue>6</issue>
          <fpage>443</fpage>
          <lpage>51; discussion 451</lpage>
          <pub-id pub-id-type="doi">10.1097/00001888-199306000-00002</pub-id>
          <pub-id pub-id-type="medline">8507309</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zerilli</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Fidler</surname>
              <given-names>BD</given-names>
            </name>
            <name name-style="western">
              <surname>Tendhar</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Assessing the impact of standardized patient encounters on students' medical history-taking skills in practice</article-title>
          <source>Am J Pharm Educ</source>
          <year>2023</year>
          <volume>87</volume>
          <issue>4</issue>
          <fpage>ajpe8989</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/36375843"/>
          </comment>
          <pub-id pub-id-type="doi">10.5688/ajpe8989</pub-id>
          <pub-id pub-id-type="medline">36375843</pub-id>
          <pub-id pub-id-type="pii">ajpe8989</pub-id>
          <pub-id pub-id-type="pmcid">PMC10159019</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cleland</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Abe</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Rethans</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>The use of simulated patients in medical education: AMEE Guide No 42</article-title>
          <source>Med Teach</source>
          <year>2009</year>
          <volume>31</volume>
          <issue>6</issue>
          <fpage>477</fpage>
          <lpage>486</lpage>
          <pub-id pub-id-type="doi">10.1080/01421590903002821</pub-id>
          <pub-id pub-id-type="medline">19811162</pub-id>
          <pub-id pub-id-type="pii">10.1080/01421590903002821</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Yan</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Guan</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Pfeiffer</surname>
              <given-names>CA</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Pan</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Standardized patient methodology in mainland China: a nationwide survey</article-title>
          <source>BMC Med Educ</source>
          <year>2019</year>
          <volume>19</volume>
          <issue>1</issue>
          <fpage>214</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-019-1630-y"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-019-1630-y</pub-id>
          <pub-id pub-id-type="medline">31208408</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-019-1630-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC6580584</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Aranda</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Monks</surname>
              <given-names>SM</given-names>
            </name>
          </person-group>
          <source>Roles and Responsibilities of the Standardized Patient Director in Medical Simulation</source>
          <year>2026</year>
          <publisher-loc>Treasure Island (FL)</publisher-loc>
          <publisher-name>StatPearls Publishing</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zeng</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Fu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Han</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Pu</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Analysis of virtual standardized patients for assessing clinical fundamental skills of medical students: a prospective study</article-title>
          <source>BMC Med Educ</source>
          <year>2024</year>
          <volume>24</volume>
          <issue>1</issue>
          <fpage>981</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-024-05982-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-024-05982-2</pub-id>
          <pub-id pub-id-type="medline">39256732</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-024-05982-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC11385815</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kononowicz</surname>
              <given-names>AA</given-names>
            </name>
            <name name-style="western">
              <surname>Woodham</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Edelbring</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Stathakarou</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Davies</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Saxena</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tudor Car</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Carlstedt-Duke</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Car</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zary</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Virtual patient simulations in health professions education: systematic review and meta-analysis by the digital health education collaboration</article-title>
          <source>J Med Internet Res</source>
          <year>2019</year>
          <volume>21</volume>
          <issue>7</issue>
          <fpage>e14676</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2019/7/e14676/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/14676</pub-id>
          <pub-id pub-id-type="medline">31267981</pub-id>
          <pub-id pub-id-type="pii">v21i7e14676</pub-id>
          <pub-id pub-id-type="pmcid">PMC6632099</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hamilton</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Molzahn</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>McLemore</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>The evolution from standardized to virtual patients in medical education</article-title>
          <source>Cureus</source>
          <year>2024</year>
          <volume>16</volume>
          <issue>10</issue>
          <fpage>e71224</fpage>
          <pub-id pub-id-type="doi">10.7759/cureus.71224</pub-id>
          <pub-id pub-id-type="medline">39525234</pub-id>
          <pub-id pub-id-type="pmcid">PMC11549952</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ellaway</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Poulton</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Fors</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>McGee</surname>
              <given-names>JB</given-names>
            </name>
            <name name-style="western">
              <surname>Albright</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Building a virtual patient commons</article-title>
          <source>Med Teach</source>
          <year>2008</year>
          <volume>30</volume>
          <issue>2</issue>
          <fpage>170</fpage>
          <lpage>174</lpage>
          <pub-id pub-id-type="doi">10.1080/01421590701874074</pub-id>
          <pub-id pub-id-type="medline">18464142</pub-id>
          <pub-id pub-id-type="pii">792859283</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>MZ</given-names>
            </name>
            <name name-style="western">
              <surname>Kwok</surname>
              <given-names>TT</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>JYH</given-names>
            </name>
          </person-group>
          <article-title>GenAI-supported virtual patients in health care education: systematic review</article-title>
          <source>J Med Internet Res</source>
          <year>2026</year>
          <volume>28</volume>
          <fpage>e82756</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2026//e82756/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/82756</pub-id>
          <pub-id pub-id-type="medline">42098926</pub-id>
          <pub-id pub-id-type="pii">v28i1e82756</pub-id>
          <pub-id pub-id-type="pmcid">PMC13152703</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Han</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Pu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Qian</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Yin</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Application of large language models in medical training evaluation-using ChatGPT as a standardized patient: multimetric assessment</article-title>
          <source>J Med Internet Res</source>
          <year>2025</year>
          <volume>27</volume>
          <fpage>e59435</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2025//e59435/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/59435</pub-id>
          <pub-id pub-id-type="medline">39742453</pub-id>
          <pub-id pub-id-type="pii">v27i1e59435</pub-id>
          <pub-id pub-id-type="pmcid">PMC11736217</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Helms</surname>
              <given-names>JT</given-names>
            </name>
            <name name-style="western">
              <surname>Crouch</surname>
              <given-names>TB</given-names>
            </name>
          </person-group>
          <article-title>Virtual patients, real conversations: ChatGPT advanced voice mode for pain communication training</article-title>
          <source>Med Teach</source>
          <year>2026</year>
          <volume>48</volume>
          <issue>3</issue>
          <fpage>357</fpage>
          <lpage>359</lpage>
          <pub-id pub-id-type="doi">10.1080/0142159X.2025.2536149</pub-id>
          <pub-id pub-id-type="medline">40705500</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dang</surname>
              <given-names>HNN</given-names>
            </name>
            <name name-style="western">
              <surname>Luong</surname>
              <given-names>TV</given-names>
            </name>
            <name name-style="western">
              <surname>Vo</surname>
              <given-names>HTH</given-names>
            </name>
            <name name-style="western">
              <surname>Nguyen</surname>
              <given-names>LTK</given-names>
            </name>
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>TN</given-names>
            </name>
            <name name-style="western">
              <surname>Pham</surname>
              <given-names>HTA</given-names>
            </name>
            <name name-style="western">
              <surname>Truong</surname>
              <given-names>HT</given-names>
            </name>
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>TT</given-names>
            </name>
            <name name-style="western">
              <surname>Hoang</surname>
              <given-names>TA</given-names>
            </name>
            <name name-style="western">
              <surname>Doan</surname>
              <given-names>TC</given-names>
            </name>
            <name name-style="western">
              <surname>Huynh</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Marwick</surname>
              <given-names>TH</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence-powered virtual standardized patients in teaching history-taking skills to medical students: A randomized controlled trial</article-title>
          <source>BMC Med Educ</source>
          <year>2026</year>
          <volume>26</volume>
          <issue>1</issue>
          <fpage>984</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-026-09305-5"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-026-09305-5</pub-id>
          <pub-id pub-id-type="medline">42063049</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-026-09305-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC13274040</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Raafat</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Harbourne</surname>
              <given-names>AD</given-names>
            </name>
            <name name-style="western">
              <surname>Radia</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Woodman</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Swales</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Saunders</surname>
              <given-names>KEA</given-names>
            </name>
          </person-group>
          <article-title>Virtual patients improve history-taking competence and confidence in medical students</article-title>
          <source>Med Teach</source>
          <year>2024</year>
          <volume>46</volume>
          <issue>5</issue>
          <fpage>682</fpage>
          <lpage>688</lpage>
          <pub-id pub-id-type="doi">10.1080/0142159X.2023.2273782</pub-id>
          <pub-id pub-id-type="medline">38084413</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Virtual standardized patients for improving clinical thinking ability training in residents: Randomized controlled trial</article-title>
          <source>JMIR Med Educ</source>
          <year>2025</year>
          <volume>11</volume>
          <fpage>e73196</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mededu.jmir.org/2025//e73196/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/73196</pub-id>
          <pub-id pub-id-type="medline">41359953</pub-id>
          <pub-id pub-id-type="pii">v11i1e73196</pub-id>
          <pub-id pub-id-type="pmcid">PMC12685284</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>Application of AI-based virtual standardized patients in physician-patient communication training: A study based on the SEGUE framework</article-title>
          <source>Front Public Health</source>
          <year>2026</year>
          <volume>14</volume>
          <fpage>1768518</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/fpubh.2026.1768518"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fpubh.2026.1768518</pub-id>
          <pub-id pub-id-type="medline">41988587</pub-id>
          <pub-id pub-id-type="pmcid">PMC13076535</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Meskó</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Prompt engineering as an important emerging skill for medical professionals: tutorial</article-title>
          <source>J Med Internet Res</source>
          <year>2023</year>
          <volume>25</volume>
          <fpage>e50638</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2023//e50638/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/50638</pub-id>
          <pub-id pub-id-type="medline">37792434</pub-id>
          <pub-id pub-id-type="pii">v25i1e50638</pub-id>
          <pub-id pub-id-type="pmcid">PMC10585440</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Eysenbach</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>The role of ChatGPT, generative language models, and artificial intelligence in medical education: a conversation with ChatGPT and a call for papers</article-title>
          <source>JMIR Med Educ</source>
          <year>2023</year>
          <volume>9</volume>
          <fpage>e46885</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mededu.jmir.org/2023//e46885/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/46885</pub-id>
          <pub-id pub-id-type="medline">36863937</pub-id>
          <pub-id pub-id-type="pii">v9i1e46885</pub-id>
          <pub-id pub-id-type="pmcid">PMC10028514</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Namvarpour</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pauwels</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Razi</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>AI-induced sexual harassment: investigating contextual characteristics and user reactions of sexual harassment by a companion chatbot</article-title>
          <source>Proc ACM Hum Comput Interact</source>
          <year>2025</year>
          <volume>9</volume>
          <issue>7</issue>
          <fpage>1</fpage>
          <lpage>28</lpage>
          <pub-id pub-id-type="doi">10.1145/3757548</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gilbert</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Philip</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Llorca</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Samalin</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Use of artificial intelligence-enhanced virtual patients in educational approaches to medical interview training: a systematic review</article-title>
          <source>BMC Med Educ</source>
          <year>2026</year>
          <volume>26</volume>
          <issue>1</issue>
          <fpage>588</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-026-08804-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-026-08804-9</pub-id>
          <pub-id pub-id-type="medline">41781962</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-026-08804-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC13067467</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <collab>National Health Commission of the People's Republic of China MoE</collab>
            <collab>Ministry of ScienceTechnology</collab>
            <collab>National Administration of Traditional Chinese Medicine</collab>
          </person-group>
          <article-title>Measures for ethical review of life science and medical research involving humans</article-title>
          <source>NHC</source>
          <year>2023</year>
          <access-date>2026-08-14</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.nhc.gov.cn/qjjys/c100016/202302/6b6e447b3edc4338856c9a652a85f44b.shtml">https://www.nhc.gov.cn/qjjys/c100016/202302/6b6e447b3edc4338856c9a652a85f44b.shtml</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Reiffel</surname>
              <given-names>JA</given-names>
            </name>
          </person-group>
          <article-title>Propensity score matching: the 'devil is in the details' where more may be hidden than you know</article-title>
          <source>Am J Med</source>
          <year>2020</year>
          <volume>133</volume>
          <issue>2</issue>
          <fpage>178</fpage>
          <lpage>181</lpage>
          <pub-id pub-id-type="doi">10.1016/j.amjmed.2019.08.055</pub-id>
          <pub-id pub-id-type="medline">31618617</pub-id>
          <pub-id pub-id-type="pii">S0002-9343(19)30853-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Stuart</surname>
              <given-names>EA</given-names>
            </name>
          </person-group>
          <article-title>Matching methods for causal inference: a review and a look forward</article-title>
          <source>Stat Sci</source>
          <year>2010</year>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <lpage>21</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/20871802"/>
          </comment>
          <pub-id pub-id-type="doi">10.1214/09-STS313</pub-id>
          <pub-id pub-id-type="medline">20871802</pub-id>
          <pub-id pub-id-type="pmcid">PMC2943670</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Austin</surname>
              <given-names>PC</given-names>
            </name>
          </person-group>
          <article-title>An introduction to propensity score methods for reducing the effects of confounding in observational studies</article-title>
          <source>Multivariate Behav Res</source>
          <year>2011</year>
          <volume>46</volume>
          <issue>3</issue>
          <fpage>399</fpage>
          <lpage>424</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.tandfonline.com/doi/10.1080/00273171.2011.568786?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1080/00273171.2011.568786</pub-id>
          <pub-id pub-id-type="medline">21818162</pub-id>
          <pub-id pub-id-type="pmcid">PMC3144483</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nabizadeh</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hajian</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sheikhan</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Rafiei</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>Prediction of academic achievement based on learning strategies and outcome expectations among medical students</article-title>
          <source>BMC Med Educ</source>
          <year>2019</year>
          <volume>19</volume>
          <issue>1</issue>
          <fpage>99</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-019-1527-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-019-1527-9</pub-id>
          <pub-id pub-id-type="medline">30953500</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-019-1527-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC6451267</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Bi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Tsui</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lai</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>A large language model digital patient system enhances ophthalmology history taking skills</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>502</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01841-6"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01841-6</pub-id>
          <pub-id pub-id-type="medline">40760042</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01841-6</pub-id>
          <pub-id pub-id-type="pmcid">PMC12322286</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cook</surname>
              <given-names>DA</given-names>
            </name>
          </person-group>
          <article-title>Creating virtual patients using large language models: scalable, global, and low cost</article-title>
          <source>Med Teach</source>
          <year>2025</year>
          <volume>47</volume>
          <issue>1</issue>
          <fpage>40</fpage>
          <lpage>42</lpage>
          <pub-id pub-id-type="doi">10.1080/0142159X.2024.2376879</pub-id>
          <pub-id pub-id-type="medline">38992981</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Anton</surname>
              <given-names>NE</given-names>
            </name>
            <name name-style="western">
              <surname>Rendina</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Hennings</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Stambro</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Stanton-Maxey</surname>
              <given-names>KJ</given-names>
            </name>
            <name name-style="western">
              <surname>Stefanidis</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Association of medical students' stress and coping skills with simulation performance</article-title>
          <source>Simul Healthc</source>
          <year>2021</year>
          <volume>16</volume>
          <issue>5</issue>
          <fpage>327</fpage>
          <lpage>333</lpage>
          <pub-id pub-id-type="doi">10.1097/SIH.0000000000000511</pub-id>
          <pub-id pub-id-type="medline">33086369</pub-id>
          <pub-id pub-id-type="pii">01266021-202110000-00005</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lipnevich</surname>
              <given-names>AA</given-names>
            </name>
            <name name-style="western">
              <surname>Mattern</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Feddock</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Formative assessment and feedback in medical education: a practical guide: AMEE Guide No. 189</article-title>
          <source>Med Teach</source>
          <year>2026</year>
          <volume>48</volume>
          <issue>6</issue>
          <fpage>921</fpage>
          <lpage>940</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.tandfonline.com/doi/10.1080/0142159X.2025.2569623?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1080/0142159X.2025.2569623</pub-id>
          <pub-id pub-id-type="medline">41134874</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gilbert</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Carnell</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lok</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Miles</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Using virtual patients to support empathy training in health care education: an exploratory study</article-title>
          <source>Simul Healthc</source>
          <year>2024</year>
          <volume>19</volume>
          <issue>3</issue>
          <fpage>151</fpage>
          <lpage>157</lpage>
          <pub-id pub-id-type="doi">10.1097/SIH.0000000000000742</pub-id>
          <pub-id pub-id-type="medline">37639216</pub-id>
          <pub-id pub-id-type="pii">01266021-202406000-00003</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Deladisma</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Cohen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Stevens</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wagner</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Lok</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Bernard</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Oxendine</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Schumacher</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Johnsen</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Dickerson</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Raij</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wells</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Duerson</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Harper</surname>
              <given-names>JG</given-names>
            </name>
            <name name-style="western">
              <surname>Lind</surname>
              <given-names>DS</given-names>
            </name>
            <collab>Association for Surgical Education</collab>
          </person-group>
          <article-title>Do medical students respond empathetically to a virtual patient?</article-title>
          <source>Am J Surg</source>
          <year>2007</year>
          <volume>193</volume>
          <issue>6</issue>
          <fpage>756</fpage>
          <lpage>760</lpage>
          <pub-id pub-id-type="doi">10.1016/j.amjsurg.2007.01.021</pub-id>
          <pub-id pub-id-type="medline">17512291</pub-id>
          <pub-id pub-id-type="pii">S0002-9610(07)00152-3</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lorié</surname>
              <given-names>Á</given-names>
            </name>
            <name name-style="western">
              <surname>Reinero</surname>
              <given-names>DA</given-names>
            </name>
            <name name-style="western">
              <surname>Phillips</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Riess</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Culture and nonverbal expressions of empathy in clinical settings: a systematic review</article-title>
          <source>Patient Educ Couns</source>
          <year>2017</year>
          <volume>100</volume>
          <issue>3</issue>
          <fpage>411</fpage>
          <lpage>424</lpage>
          <pub-id pub-id-type="doi">10.1016/j.pec.2016.09.018</pub-id>
          <pub-id pub-id-type="medline">27693082</pub-id>
          <pub-id pub-id-type="pii">S0738-3991(16)30446-3</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Maicher</surname>
              <given-names>KR</given-names>
            </name>
            <name name-style="western">
              <surname>Zimmerman</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Wilcox</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Liston</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Cronau</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Macerollo</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Jaffe</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>White</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Fosler-Lussier</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Schuler</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Way</surname>
              <given-names>DP</given-names>
            </name>
            <name name-style="western">
              <surname>Danforth</surname>
              <given-names>DR</given-names>
            </name>
          </person-group>
          <article-title>Using virtual standardized patients to accurately assess information gathering skills in medical students</article-title>
          <source>Med Teach</source>
          <year>2019</year>
          <volume>41</volume>
          <issue>9</issue>
          <fpage>1053</fpage>
          <lpage>1059</lpage>
          <pub-id pub-id-type="doi">10.1080/0142159X.2019.1616683</pub-id>
          <pub-id pub-id-type="medline">31230496</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ning</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ong</surname>
              <given-names>JCL</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DSW</given-names>
            </name>
            <name name-style="western">
              <surname>Tham</surname>
              <given-names>YC</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>TY</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>How can artificial intelligence transform the training of medical students and physicians?</article-title>
          <source>Lancet Digit Health</source>
          <year>2025</year>
          <volume>7</volume>
          <issue>10</issue>
          <fpage>100900</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2589-7500(25)00082-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.landig.2025.100900</pub-id>
          <pub-id pub-id-type="medline">41047321</pub-id>
          <pub-id pub-id-type="pii">S2589-7500(25)00082-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sawyer</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>White</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zaveri</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Ades</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>French</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Anderson</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Auerbach</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Johnston</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Kessler</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Learn, see, practice, prove, do, maintain: An evidence-based pedagogical framework for procedural skill training in medicine</article-title>
          <source>Acad Med</source>
          <year>2015</year>
          <volume>90</volume>
          <issue>8</issue>
          <fpage>1025</fpage>
          <lpage>1033</lpage>
          <pub-id pub-id-type="doi">10.1097/ACM.0000000000000734</pub-id>
          <pub-id pub-id-type="medline">25881645</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hawkins</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Younan</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Fyfe</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Parekh</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>McKeown</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Exploring why medical students still feel underprepared for clinical practice: a qualitative analysis of an authentic on-call simulation</article-title>
          <source>BMC Med Educ</source>
          <year>2021</year>
          <volume>21</volume>
          <issue>1</issue>
          <fpage>165</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-021-02605-y"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-021-02605-y</pub-id>
          <pub-id pub-id-type="medline">33731104</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-021-02605-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC7972243</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref40">
        <label>40</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Burgess</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>van Diggele</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Roberts</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mellis</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Tips for teaching procedural skills</article-title>
          <source>BMC Med Educ</source>
          <year>2020</year>
          <volume>20</volume>
          <issue>Suppl 2</issue>
          <fpage>458</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-020-02284-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-020-02284-1</pub-id>
          <pub-id pub-id-type="medline">33272273</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-020-02284-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC7712522</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref41">
        <label>41</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Abraham</surname>
              <given-names>RM</given-names>
            </name>
            <name name-style="western">
              <surname>Singaram</surname>
              <given-names>VS</given-names>
            </name>
          </person-group>
          <article-title>Using deliberate practice framework to assess the quality of feedback in undergraduate clinical skills training</article-title>
          <source>BMC Med Educ</source>
          <year>2019</year>
          <volume>19</volume>
          <issue>1</issue>
          <fpage>105</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-019-1547-5"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-019-1547-5</pub-id>
          <pub-id pub-id-type="medline">30975213</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-019-1547-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC6460682</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref42">
        <label>42</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Izquierdo-Condoy</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>Arias-Intriago</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Tello-De-la-Torre</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Busch</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Ortiz-Prado</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Generative artificial intelligence in medical education: enhancing critical thinking or undermining cognitive autonomy?</article-title>
          <source>J Med Internet Res</source>
          <year>2025</year>
          <volume>27</volume>
          <fpage>e76340</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2025//e76340/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/76340</pub-id>
          <pub-id pub-id-type="medline">41183320</pub-id>
          <pub-id pub-id-type="pii">v27i1e76340</pub-id>
          <pub-id pub-id-type="pmcid">PMC12624298</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref43">
        <label>43</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Pham</surname>
              <given-names>TD</given-names>
            </name>
            <name name-style="western">
              <surname>Karunaratne</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Exintaris</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Lay</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Yuriev</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>The impact of generative AI on health professional education: a systematic review in the context of student learning</article-title>
          <source>Med Educ</source>
          <year>2025</year>
          <volume>59</volume>
          <issue>12</issue>
          <fpage>1280</fpage>
          <lpage>1289</lpage>
          <pub-id pub-id-type="doi">10.1111/medu.15746</pub-id>
          <pub-id pub-id-type="medline">40533396</pub-id>
          <pub-id pub-id-type="pmcid">PMC12686775</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref44">
        <label>44</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Weidener</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Fischer</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence teaching as part of medical education: Qualitative analysis of expert interviews</article-title>
          <source>JMIR Med Educ</source>
          <year>2023</year>
          <volume>9</volume>
          <fpage>e46428</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mededu.jmir.org/2023//e46428/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/46428</pub-id>
          <pub-id pub-id-type="medline">36946094</pub-id>
          <pub-id pub-id-type="pii">v9i1e46428</pub-id>
          <pub-id pub-id-type="pmcid">PMC10167581</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref45">
        <label>45</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lenihan</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Three effective, efficient, and easily implementable ways to integrate A.I. into medical education</article-title>
          <source>Cureus</source>
          <year>2023</year>
          <volume>15</volume>
          <issue>10</issue>
          <fpage>e47204</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37854479"/>
          </comment>
          <pub-id pub-id-type="doi">10.7759/cureus.47204</pub-id>
          <pub-id pub-id-type="medline">37854479</pub-id>
          <pub-id pub-id-type="pmcid">PMC10581027</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref46">
        <label>46</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>D'Alessandro</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Adhikari</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Goff</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Bargagli-Stoffi</surname>
              <given-names>FJ</given-names>
            </name>
            <name name-style="western">
              <surname>Santacatterina</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Modern causal inference approaches to improve power for subgroup analysis in randomized controlled trials</article-title>
          <source>Stat Med</source>
          <year>2026</year>
          <volume>45</volume>
          <issue>3-5</issue>
          <fpage>e70436</fpage>
          <pub-id pub-id-type="doi">10.1002/sim.70436</pub-id>
          <pub-id pub-id-type="medline">41700655</pub-id>
          <pub-id pub-id-type="pmcid">PMC13213542</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref47">
        <label>47</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Merkebu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Samuel</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Humanizing AI training for health professions educators</article-title>
          <source>Med Teach</source>
          <year>2026</year>
          <volume>48</volume>
          <issue>3</issue>
          <fpage>360</fpage>
          <lpage>363</lpage>
          <pub-id pub-id-type="doi">10.1080/0142159X.2025.2522237</pub-id>
          <pub-id pub-id-type="medline">40632734</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref48">
        <label>48</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kirkman</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Simmonds</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pook</surname>
              <given-names>CJ</given-names>
            </name>
            <name name-style="western">
              <surname>Haas-Heger</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>Overcoming barriers for the implementation of simulation-based education within lower-resource settings: a medical student perspective</article-title>
          <source>J Clin Nurs</source>
          <year>2023</year>
          <volume>32</volume>
          <issue>11-12</issue>
          <fpage>2941</fpage>
          <lpage>2942</lpage>
          <pub-id pub-id-type="doi">10.1111/jocn.16063</pub-id>
          <pub-id pub-id-type="medline">34585456</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
