<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id><journal-id journal-id-type="publisher-id">mededu</journal-id><journal-id journal-id-type="index">20</journal-id><journal-title>JMIR Medical Education</journal-title><abbrev-journal-title>JMIR Med Educ</abbrev-journal-title><issn pub-type="epub">2369-3762</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v12i1e95039</article-id><article-id pub-id-type="doi">10.2196/95039</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Scaffolded AI-Supported Problem-Based Learning for Medical Interns: Exploratory Retrospectively Registered Randomized Controlled Evaluation With a Voluntary Feasibility Follow-Up</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Zhao</surname><given-names>Zhengqi</given-names></name><degrees>BEng</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Wu</surname><given-names>Baijing</given-names></name><degrees>BM</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tian</surname><given-names>Shuyuan</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hu</surname><given-names>Haigen</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Hao</surname><given-names>Pengyi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Computer Science and Technology, Zhejiang University of Technology</institution><addr-line>No. 288 Liuhe Road, Xihu District</addr-line><addr-line>Hangzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff2"><institution>Zhejiang Institute of Artificial Intelligence</institution><addr-line>Hangzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Ultrasound, Tongde Hospital Affiliated to Zhejiang Chinese Medical University</institution><addr-line>No. 234 Gucui Road</addr-line><addr-line>Hangzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Lesselroth</surname><given-names>Blake</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Golamari</surname><given-names>Bala Vinay Kumar</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Rubinstein</surname><given-names>Beny</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Arpaci</surname><given-names>Ibrahim</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Pengyi Hao, PhD, Department of Computer Science and Technology, Zhejiang University of Technology, No. 288 Liuhe Road, Xihu District, Hangzhou, Zhejiang, 310023, China, 86 158 6913 2329; <email>haopy@zjut.edu.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>3</day><month>9</month><year>2026</year></pub-date><volume>12</volume><elocation-id>e95039</elocation-id><history><date date-type="received"><day>10</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>03</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>11</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Zhengqi Zhao, Baijing Wu, Shuyuan Tian, Haigen Hu, Pengyi Hao. Originally published in JMIR Medical Education (<ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org">https://mededu.jmir.org</ext-link>), 3.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org/">https://mededu.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mededu.jmir.org/2026/1/e95039"/><abstract><sec><title>Background</title><p>High-quality problem-based learning (PBL) during internship is resource-intensive and difficult to scale without consistent facilitation. Although generative AI is increasingly used in health professions education, many applications remain on-demand answer tools that may not reproduce core PBL processes.</p></sec><sec><title>Objective</title><p>This study aimed to develop and evaluate the Multiagent PBL Environment for Clinical Reasoning (MAPLE-CR), an AI-supported environment that uses generative AI as a process-oriented scaffold rather than an answer-delivery aid. The system supports cognitive mechanisms through reasoning prompts, social-interactional mechanisms through simulated tutor and peer roles, and regulatory mechanisms through structured workflows and feedback loops. We separately evaluated implementation and repeated-use feasibility, including learner experience, short-term outcomes versus self-study, and preliminary framework-aligned process evidence.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a 2-stage study. In an exploratory randomized evaluation (N=52; intervention: n=26; control: n=26), interns completed parallel pretests and posttests around a standardized case. The intervention group engaged in asynchronous, scaffolded PBL in MAPLE-CR, whereas controls completed case-matched self-study using materials derived from the same case and learning objectives. We summarized transcript-derived process dimensions and postsession learner experience in the intervention group. In a voluntary follow-up, 18 participants completed 1 MAPLE-CR case per week for 4 additional weeks to examine repeated use feasibility and patterns across novel cases.</p></sec><sec sec-type="results"><title>Results</title><p>Baseline pretest scores were comparable between groups (<italic>P</italic>=.56). The intervention group achieved higher posttest scores than controls (Hodges-Lehmann median difference 8.25 points, 95% CI 6.60-11.60; Cliff &#x03B4;=0.506, 95% CI 0.21&#x2010;0.76; <italic>P</italic>=.001) and greater score gains (median difference 6.60 points, 95% CI 1.60-14.95; <italic>P</italic>=.01). On a 0 to 5 scale, mean scores were highest for knowledge accuracy (4.4) and clinical reasoning (CR, 3.7), whereas active participation averaged 2.1, and interaction-oriented dimensions were more variable. Means for 12 positively worded questionnaire items ranged from 4.19 to 4.62, supporting high satisfaction, involvement, perceived support, and self-efficacy; open-ended responses contextualized perceived strengths and improvement needs. In the voluntary follow-up (n=18), weekly cross-case use was feasible; median within-session gains ranged from 13.4 to 20.0 points across 5 attempts, with marked increases in CR and aggregate scores from attempts 1 to 2 but statistically uncertain later trajectories.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>MAPLE-CR was feasible and showed high learner acceptability. Its use was associated with larger short-term CR knowledge gains than case-matched self-study, while transcript analyses provided preliminary framework-aligned process evidence. However, the post hoc sensitivity analysis did not establish prospective power sufficiency, and the confidence interval could not exclude smaller educationally meaningful effects. These findings suggest that AI-supported, process-oriented PBL may provide a scalable formative supplement for CR practice when facilitator capacity and small-group scheduling are constrained, but the study did not isolate specific scaffolding effects. Further research should confirm these findings using larger, prospectively powered trials with stronger active comparators and longer-term outcome measures.</p></sec><sec><title>Trial Registration</title><p>Chinese Clinical Trial Registry ChiCTR2500115799; https://tinyurl.com/353y6c92</p></sec></abstract><kwd-group><kwd>problem-based learning</kwd><kwd>clinical reasoning</kwd><kwd>scaffolding</kwd><kwd>artificial intelligence</kwd><kwd>large language models</kwd><kwd>educational technology</kwd><kwd>randomized controlled trial</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Medical interns must translate biomedical knowledge into timely clinical decisions under uncertainty. Problem-based learning (PBL) supports this transition by engaging learners in case analysis, hypothesis generation, interpretation of evolving information, and iterative revision of explanations and differential diagnoses [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Its educational value depends not only on exposure to cases but also on learners&#x2019; active articulation of reasoning, identification of knowledge gaps, self-directed study, and integration of new understanding [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>Despite these benefits, scaling high-quality PBL in routine clinical curricula remains difficult. Effective PBL typically depends on skilled facilitation, small-group coordination, and protected instructional time, all of which are resource-intensive and difficult to sustain consistently during internship [<xref ref-type="bibr" rid="ref6">6</xref>]. In practice, the challenge is not only increasing access to PBL-like activities but also preserving the pedagogical processes that make PBL distinctive. Its educational value depends on collaborative sensemaking, externalization of reasoning, and facilitator-guided progression through a structured case process [<xref ref-type="bibr" rid="ref7">7</xref>]. When these enabling processes are weak or absent, PBL can deteriorate into superficial discussion or premature answer-seeking rather than disciplined reasoning.</p><p>Against these implementation constraints, generative AI, especially large language model (LLM) systems capable of flexible multiturn dialogue, has emerged as a potential way to expand learning opportunities in time- and resource-constrained clinical settings [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref12">12</xref>]. Early work in health professions education has explored generative AI for functions such as content support, explanation, feedback, and learner assistance [<xref ref-type="bibr" rid="ref13">13</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. Several studies have also begun to examine AI-supported learning in PBL-related contexts, with encouraging findings for learner support, idea generation, and just-in-time explanation [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. However, in many of these implementations, AI functions primarily as a consultation aid or question-answering assistant embedded within the learning process, rather than as support for orchestrating the PBL process itself.</p><p>This distinction matters because question answering is not equivalent to PBL. PBL is not only a matter of information access [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]; it is a structured learning process in which learners are expected to generate and justify hypotheses, respond to new information, engage with alternative perspectives, and move through recognizable phases of inquiry [<xref ref-type="bibr" rid="ref18">18</xref>]. Its benefits also depend on feedback and continuity across cases, so that learners can reflect on performance, recalibrate reasoning, and improve over time [<xref ref-type="bibr" rid="ref19">19</xref>]. Accordingly, simply providing answers or ad hoc explanations is unlikely to reproduce the core pedagogical functions of PBL. The specific gap addressed in this study is therefore not the absence of AI assistance within case learning but the limited evidence on AI-supported PBL environments designed to scaffold multiple phases of the PBL cycle as a facilitator-like process [<xref ref-type="bibr" rid="ref20">20</xref>], integrating presession preparation, in-session reasoning and peer interaction, postsession feedback, and subsequent practice rather than supporting only isolated question answering or explanation.</p><p>In this study, we adopt a process-oriented view of scaffolding for AI-supported PBL. We use scaffolding to mean structured support embedded in the learning workflow to help learners perform reasoning, interaction, and reflection activities that would otherwise require sustained tutor facilitation. We use orchestration to refer to the coordination of scaffolded activities across the full PBL cycle, including presession preparation, case discussion, postsession feedback, and subsequent practice. Within this theoretical framework, MAPLE-CR (Multiagent PBL Environment for Clinical Reasoning) was designed around 3 complementary forms of process-oriented scaffolding. First, cognitive scaffolding prompts learners to generate hypotheses, justify them with case evidence, revise interpretations as new information emerges, and synthesize key clinical points [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Second, social-interactional scaffolding uses simulated tutor and peer roles to sustain dialogue, questioning, turn-taking, peer-responsive reasoning, and externalization of thinking rather than a purely monologic learner-AI exchange [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. Third, regulatory scaffolding structures the workflow, monitors objective coverage, manages phase transitions, and embeds feedback loops across repeated sessions to support reflection, calibration, and progressive improvement [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. This framework treats scalability not simply as replacing human labor with automation but as preserving the cognitive, dialogic, and feedback-regulatory functions that make PBL educationally meaningful.</p><p>Guided by this framework, we developed MAPLE-CR, an AI-supported learning environment for continuous, scaffolded PBL practice among medical interns. It was intended to provide a structured yet flexible environment for clinical reasoning (CR) practice when in-person facilitation and repeated small-group sessions are difficult to sustain. Accordingly, this study aimed to evaluate MAPLE-CR across three distinct evidentiary domains: (1) feasibility of implementation and repeated cross-case use, together with learner acceptability and perceived educational value; (2) short-term learning-outcome differences relative to case-matched self-study in an exploratory randomized evaluation; and (3) preliminary framework-aligned process evidence based on transcript-derived formative dimensions, with expert ratings used as an external reference (<xref ref-type="table" rid="table1">Table 1</xref> summarizes this mapping and related expert-rating checks). The voluntary follow-up was interpreted descriptively as evidence of repeated-use feasibility rather than longitudinal learning effectiveness.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Intended mapping of Multiagent Problem-based Learning Environment for Clinical Reasoning (MAPLE-CR) components to scaffolding functions, formative dimensions, and expert-rating validation.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Scaffolding function</td><td align="left" valign="bottom">MAPLE-CR components and intended PBL<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> support</td><td align="left" valign="bottom">Framework-aligned formative dimensions and expert-rating validation</td></tr></thead><tbody><tr><td align="left" valign="top">Cognitive scaffolding</td><td align="left" valign="top">Learning objectives, sequential case scenarios, and tutor prompts for hypothesis generation, evidence justification, revision, and synthesis. These components were intended to elicit diagnostic and management reasoning aligned with case objectives.</td><td align="left" valign="top">CR<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>; KA<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>; expert ratings for CR and KA</td></tr><tr><td align="left" valign="top">Social-interactional scaffolding</td><td align="left" valign="top">Tutor and peer roles, peer prompts and counterpoints, and turn-taking support. These components were intended to sustain multivoice discussion and externalization of reasoning.</td><td align="left" valign="top">AP<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>; CC<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup>; expert ratings for AP and CC</td></tr><tr><td align="left" valign="top">Regulatory scaffolding</td><td align="left" valign="top">Case recommendation, pretest-informed learner profile, objective-coverage monitoring, phase transitions, summaries, feedback, and updated learning record. These components were intended to structure reflection, calibration, and repeated practice.</td><td align="left" valign="top">SR<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup>; expert ratings for SR</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>PBL: problem-based learning.</p></fn><fn id="table1fn2"><p><sup>b</sup>CR: clinical reasoning.</p></fn><fn id="table1fn3"><p><sup>c</sup>KA: knowledge accuracy.</p></fn><fn id="table1fn4"><p><sup>d</sup>AP: active participation.</p></fn><fn id="table1fn5"><p><sup>e</sup>CC: communication and collaboration.</p></fn><fn id="table1fn6"><p><sup>f</sup>SR: summarization and reflection.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>MAPLE-CR</title><p>MAPLE-CR is a web-based, LLM-supported PBL environment for medical interns; an overview of the system structure and learning workflow is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. In implementation terms, an &#x201C;agent&#x201D; refers to an LLM-guided software role rather than a human participant, consistent with recent work on LLM-based agents and multiagent collaboration [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. MAPLE-CR instantiates tutor, peer, evaluator, and advisor functions as distinct agents and uses multiagent orchestration to coordinate role interactions and workflow sequencing. Learners interact with MAPLE-CR through 4 connected modules: recommendation, testing, PBL training, and assessment with feedback, all supported by a shared data and knowledge layer containing cases, learning objectives, examination items, learner records, discussion transcripts, formative scores, and knowledge graph resources.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Multiagent Problem-based Learning Environment for Clinical Reasoning closed-loop problem-based learning workflow. Learners select a recommended case, complete a brief pretest, engage in scenario-based small-group discussion with a tutor role and simulated peers, and then complete a posttest. Transcript-based formative feedback is generated across 5 dimensions. Session outputs update the learner&#x2019;s trajectory to inform subsequent case recommendations, enabling repeated, scaffolded practice. AP: active participation; CC: communication and collaboration; CR: clinical reasoning; KA: knowledge accuracy; PBL: problem-based learning; SR: summarization and reflection.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95039_fig01.png"/></fig><p>The recommendation and testing modules initiate each learning cycle. The recommendation module uses the learner&#x2019;s prior learning trajectory, defined as the longitudinal record of completed cases and assessment history, to generate a ranked list of candidate cases. The learner selects 1 recommended case, preserving learner choice while supporting continuity across repeated practice. The testing module then administers brief case-linked pretests and posttests aligned with the case learning objectives and target knowledge points; the pretest summarizes baseline case-relevant understanding, whereas the posttest assesses immediate knowledge performance after the PBL session.</p><p>The PBL training module is the objective-guided learning environment. Cases are organized around prespecified learning objectives and sequential scenarios. Within each scenario, the tutor agent presents case information, monitors objective coverage, and supports transition to summary phases. Peer-agent personas combine fixed role constraints with learner-specific ability calibration: the fixed prompt defines the peer as a medical learner who participates in discussion, asks questions, and responds to others, whereas the dynamic component adjusts the peer agent&#x2019;s ability profile according to the learner&#x2019;s pretest-derived baseline.</p><p>The assessment with feedback module combines summative and formative information. Summative feedback includes the pretest score, posttest score, and score change. Formative assessment is generated from the discussion transcript across 5 PBL-relevant dimensions aligned with the scaffolding framework: active participation (AP), CR, knowledge accuracy (KA), communication and collaboration (CC), and summarization and reflection (SR). AP is computed algorithmically from learner interaction frequency, whereas CR, KA, CC, and SR are scored by an evaluator agent using dimension-specific prompts and the full discussion context. CR and KA additionally use retrieval from the structured medical knowledge graph as a clinical reference [<xref ref-type="bibr" rid="ref14">14</xref>]. Evaluator-agent scores were constrained to a 0 to 5 scale in 0.5-point increments and accompanied by brief dimension-specific justifications. The resulting feedback summarizes strengths and areas for improvement and updates the learner record for subsequent case recommendations.</p><p>MAPLE-CR was implemented as a remotely accessed web platform, with a Streamlit interface deployed on a cloud server. Model inference used the Qwen-flash API (qwen-flash-2025-07-28) [<xref ref-type="bibr" rid="ref27">27</xref>], with temperature set to 0.7. Only pseudonymized educational content was transmitted to the Qwen API for model inference; direct personal identifiers, account credentials, and study-specific participant identifiers were excluded from API requests. The user interface for the training module is shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>. Further details on API data flow, access control, data retention, and consent are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> provides a video demonstration of the system, and <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> provides additional implementation details, including module interactions, system configuration, peer-agent calibration, and agent prompts.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>User interface of the problem-based learning (PBL) training module in Multiagent PBL Environment for Clinical Reasoning. The screenshot shows a later management-oriented scenario in which the tutor agent presents staged case information and learning objectives to structure discussion around the selected case. Within this objective-guided format, the peer agent initiates discussion and the learner contributes through text input. BP: blood pressure; HR: heart rate; LDL: low-density lipoprotein.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95039_fig02.png"/></fig></sec><sec id="s2-2"><title>Study Design</title><p>We conducted an exploratory randomized controlled educational evaluation followed by a voluntary 4-week repeated-use feasibility follow-up. Reporting was guided by the CONSORT-eHEALTH (Consolidated Standards of Reporting Trials of Electronic and Mobile Health Applications and Online Telehealth) checklist (<xref ref-type="supplementary-material" rid="app6">Checklist 1</xref>).</p><p>Fifty-two first-year postgraduate medical interns (resident interns) were recruited from Tongde Hospital Affiliated to Zhejiang Chinese Medical University, a tertiary teaching hospital in Zhejiang Province, China. All had prior face-to-face PBL experience and reported previous use of GPT-style chatbots, but none had used MAPLE-CR. Participants were randomized 1:1 to the intervention or control group using a computer-generated sequence prepared by a researcher not otherwise involved in the trial. Participant recruitment and enrollment took place from December 6 to 15, 2025, and the randomized educational evaluation was conducted on December 19, 2025, before trial registration; therefore, the trial was retrospectively registered.</p><p>MAPLE-CR system and code development, including backend model integration and platform deployment, began in August 2025 and was followed by internal testing and debugging by the research team before participant recruitment. These deployment and development activities were limited to internal testing and debugging and did not involve participant enrollment, randomization, or collection of trial outcome data.</p><p>All participants completed the pretest concurrently in an on-site computer laboratory. To reduce contamination, groups were seated in separate classrooms. Research team members monitored the classrooms to maintain timing and ensure access to the assigned materials or MAPLE-CR platform; they did not provide clinical teaching, facilitate discussion, answer case-related questions, or collect additional outcome data beyond documenting attendance and session completion. Allocation was concealed until completion of the pretest; participant blinding after allocation was not feasible. Pretest, posttest, questionnaire, and follow-up records were linked using study-specific participant identifiers; analyses were conducted on deidentified datasets without names or other direct personal identifiers. To standardize exposure, all participants used the same standardized PBL case, and the recommendation function was therefore not used in the randomized component. The case focused on CR for a right middle-lobe lung mass ultimately diagnosed as lung adenocarcinoma. The case materials, learning objectives, and associated pretest and posttest items were developed and reviewed by 3 domain experts. Pretest and posttest were parallel forms based on a shared blueprint (see <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>); experts reviewed item difficulty, learning-objective coverage, and consistency between the case materials and assessment items, resolving discrepancies by consensus. Each form comprised 30 single-best-answer multiple-choice items scored automatically against a fixed answer key as the percentage of correct responses (range 0&#x2010;100), without using group-assignment information; all items had to be completed before submission. Blueprint development and expert review preceded the trial. To provide supplementary evidence for the assessment instruments, we conducted post hoc psychometric checks using actual item-level responses and independent 5-point expert difficulty ratings for all pretest and posttest items. These analyses supported acceptable internal consistency, item discrimination, and comparable expert-rated difficulty between forms; detailed methods and results are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>Immediately after the pretest, participants in the intervention group received a standardized 40-minute orientation to MAPLE-CR before individually completing up to 2 hours of PBL training. The orientation introduced the platform workflow, interface operations, tutor and peer roles, and basic expectations for participating in the AI-supported PBL discussion. It did not include case-specific diagnostic or management instruction, posttest answers, or substantive teaching on the target case. Because the orientation was intended to support learners&#x2019; engagement with the platform, it was considered part of the MAPLE-CR intervention package. If discussion ended early, participants reviewed the transcript until the session ended. A 2-hour case-matched self-study condition was selected as a feasible independent-learning comparator for this exploratory evaluation, using expert-reviewed materials to standardize case content and learning objectives without matching MAPLE-CR&#x2019;s interaction, feedback, novelty, or procedural guidance. Both groups then completed the posttest concurrently. Because the 40-minute orientation had no time-matched counterpart in the control condition, the conditions were not equivalent in total structured exposure. Accordingly, the comparison estimates the effect of the integrated MAPLE-CR intervention package under the present implementation rather than the isolated effect of its AI-PBL scaffolding components. In the intervention group, MAPLE-CR generated transcript-based formative feedback only after the posttest; no comparable feedback was provided to controls.</p><p>After the session, participants completed a postsession questionnaire adapted from a previously published simulation-based training quality assurance instrument [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Because the original instrument was designed for simulation-based training rather than AI-supported PBL, item wording and domains were modified to fit the MAPLE-CR context. Items were rated on a 5-point Likert scale from strongly disagree to strongly agree.</p><p>Following the controlled study, intervention participants were invited to join a voluntary 4-week follow-up, completing 1 MAPLE-CR case per week with the full closed-loop workflow (recommendation, testing, PBL training, assessment, and feedback). At follow-up completion, participants provided open-ended feedback on their repeated-use experiences. The voluntary follow-up was conducted from December 22, 2025, to January 19, 2026.</p><p>No important changes to study procedures, intervention content, or platform functionality occurred after trial commencement, and no major technical failures or downtime affecting study delivery were observed.</p></sec><sec id="s2-3"><title>Sample Size and Power Considerations</title><p>No formal a priori sample size calculation was conducted. The available sample size was determined by the fixed intern cohort, the number of eligible participants who could be recruited, and the feasibility of conducting a synchronized on-site educational experiment. Accordingly, the randomized component should be interpreted as an exploratory proof-of-concept evaluation rather than a fully powered confirmatory efficacy trial. To clarify the detectable magnitude under the available sample, we conducted a post hoc sample size sensitivity analysis based on the observed posttest effect. The observed Cliff &#x03B4; was 0.506, corresponding to an area under the curve of (&#x03B4;+1)/2=0.753 and an approximate Cohen <italic>d</italic> of 0.97. With 26 participants per group and 2-sided &#x03B1;=.05, this observed effect was larger than the standardized mean difference corresponding approximately to 80% power (<italic>d</italic>&#x2248;0.78) and 90% power (<italic>d</italic>&#x2248;0.90). However, the lower bound of the 95% CI for Cliff &#x03B4; (0.21) corresponds to an approximate <italic>d</italic> of 0.38, indicating that the effect-size estimate remains imprecise and that smaller but educationally meaningful effects cannot be excluded. This analysis is therefore presented only as a descriptive sensitivity analysis and does not substitute for a prospective sample size calculation.</p></sec><sec id="s2-4"><title>Registration and Analysis Status</title><p>The registration was completed retrospectively and specified broad outcome domains related to CR ability and knowledge mastery; it did not prospectively specify the operational outcomes, hypotheses, statistical analysis plan, or voluntary follow-up measures used in this educational evaluation. The pretest and posttest knowledge assessments, postsession questionnaire, transcript-based formative assessment, and voluntary repeated-use follow-up were incorporated into the study procedures; however, because registration followed the randomized evaluation, none constituted a prospectively registered confirmatory outcome, and their analyses are interpreted as exploratory or descriptive. Expert-rating convergence, session-level correlations, post hoc psychometric checks, the questionnaire response-pattern sensitivity analysis, and sample size sensitivity calculations were not specified in the registration and are reported as exploratory, sensitivity, or post hoc analyses, as applicable.</p></sec><sec id="s2-5"><title>Statistical Analyses</title><p>All 52 randomized participants completed the pretest and posttest and were analyzed as assigned under an intention-to-treat approach; intervention questionnaires and transcripts were complete, and no imputation was required. We used primarily nonparametric methods because of the modest sample size and the ordinal nature of questionnaire data. Baseline group characteristics and between-group outcomes (pretest, posttest, and score gain) were compared using Fisher exact tests or Mann-Whitney <italic>U</italic> tests, as appropriate. Within-group pretest-posttest changes were assessed using Wilcoxon signed-rank tests. For repeated-use outcomes across 5 attempts, we used Friedman tests for repeated within-subject comparisons. We report effect sizes and 95% CIs where applicable (eg, Cliff &#x03B4;, Hodges-Lehmann median differences, or bootstrap CIs). All tests were 2-sided with &#x03B1;=.05. Among intervention participants only, transcript-based formative scores were summarized for the 26 MAPLE-CR sessions. Convergent validity was examined using Spearman rank correlations (&#x03C1;) and mean absolute error (MAE) against mean expert ratings, and exploratory session-level associations within the intervention group were examined using 2-sided Spearman rank correlations (&#x03C1;), given the small sample size and the bounded or ordinal nature of several variables. Because no formal a priori sample size calculation was conducted, we supplemented the randomized outcome analysis with a post hoc sample size sensitivity analysis based on the observed posttest Cliff &#x03B4;, converting Cliff &#x03B4; to area under the curve and to an approximate standardized mean difference. Interrater reliability of the expert reference standard was assessed for each PBL dimension using a 2-way mixed-effects absolute-agreement intraclass correlation coefficient (ICC). Because convergent-validity analyses used mean expert ratings, we report ICC(A,3), the reliability of the 3-rater mean.</p></sec><sec id="s2-6"><title>Ethical Considerations</title><p>The study was conducted in accordance with the Declaration of Helsinki and approved by the Ethics Committee of Tongde Hospital of Zhejiang Province (KTSC2025080). The trial was retrospectively registered with the Chinese Clinical Trial Registry (ChiCTR2500115799; registered on December 31, 2025). Participation was voluntary, and all participants provided informed consent before study participation, including consent to the cloud-based AI processing and research use of pseudonymized dialogue, learning, and assessment data. Data were deidentified before analysis. This was a minimal-risk educational study using simulated cases.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Results of the Randomized Controlled Study</title><p>Fifty-two interns were randomized (intervention: n=26, control: n=26), and 18 completed the voluntary follow-up (<xref ref-type="fig" rid="figure3">Figure 3</xref>). Baseline sex and age were comparable between the intervention and control groups (female/male: 18/8 vs 20/6, Fisher exact test, <italic>P</italic>=.76; median age 23, IQR 23-24 vs 24, IQR 23-24, Mann-Whitney <italic>U</italic> test, <italic>P</italic>=.31). The follow-up subgroup did not differ from the intervention group in sex (female/male: 11/7 vs 18/8, Fisher exact test <italic>P</italic>=.75) or age (median 23, IQR 23-24 vs 23, IQR 23-24, Mann-Whitney <italic>U</italic> test, <italic>P</italic>=.78).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Study design and participant flow through the randomized evaluation and voluntary feasibility follow-up. Fifty-two interns were randomized 1:1 to Multiagent Problem-based Learning Environment for Clinical Reasoning (MAPLE-CR) or case-matched self-study and were included in the randomized analysis. A voluntary subset of MAPLE-CR participants subsequently completed repeated follow-up sessions across additional cases. PBL: problem-based learning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95039_fig03.png"/></fig><p>Baseline pretest scores were comparable between the intervention and control groups (Mann-Whitney <italic>U</italic> test, <italic>P</italic>=.56; <xref ref-type="fig" rid="figure4">Figure 4</xref>), suggesting no meaningful baseline imbalance in initial knowledge. After the session, the intervention group achieved higher posttest scores than the control group (Mann-Whitney <italic>U</italic> test, <italic>P</italic>=.001), with a between-group median difference of 8.25 (95% CI 6.60&#x2010;11.60) points. The corresponding effect size was Cliff &#x03B4;=0.506 (95% CI 0.21-0.76), indicating a distributional shift toward higher posttest scores in the intervention group. Consistently, the intervention group demonstrated a greater score gain (posttest minus pretest) than the control group (Mann-Whitney <italic>U</italic> test, <italic>P=</italic>.01), with a between-group median difference of 6.60 (95% CI 1.60&#x2010;14.95) points. The CI for this gain difference was wide, indicating uncertainty in the magnitude of the incremental benefit. Within-group analyses indicated that the control group also improved from pretest to posttest (Wilcoxon signed-rank test, <italic>P=</italic>.006), with a median gain of +6.70 (95% CI 0.10-10.05) points. This pattern is consistent with a measurable learning effect under the self-study condition rather than mere test-retest fluctuation. Overall, between-group separation was observed for posttest performance and gain under the present comparator condition. Because the control group also improved, this difference should be interpreted as a MAPLE-CR-associated short-term difference rather than evidence that the effect was attributable specifically to PBL scaffolding mechanisms.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Group distributions of pretest scores, posttest scores, and score gain. Scores are on a 0 to 100 scale. Data were analyzed using the Mann-Whitney <italic>U</italic> test. *<italic>P</italic>&#x003C;.05. **<italic>P</italic>&#x003C;.01. ns: not significant.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95039_fig04.png"/></fig><p>The observed posttest effect was large in this sample. In the sample size sensitivity analysis, the observed effect magnitude exceeded the approximate large-effect threshold detectable with 26 participants per group under 2-sided &#x03B1;=.05. However, the confidence interval around Cliff &#x03B4; was wide, and its lower bound corresponded to a substantially smaller standardized effect. The current sample therefore cannot exclude smaller but still educationally meaningful effects. Because this was a descriptive sensitivity analysis, these findings should be interpreted as evidence of a large short-term difference under the present study conditions, rather than as definitive efficacy evidence.</p><p><xref ref-type="table" rid="table2">Table 2</xref> summarizes postsession questionnaire responses following the MAPLE-CR intervention, presenting full-sample results as the main descriptive analysis and filtered results as a sensitivity analysis. Across the 12 positively worded items (Q1-Q12), full-sample ratings were consistently favorable, with means ranging from 4.19 to 4.62, indicating high satisfaction, sustained involvement, strong perceived support, and enhanced self-efficacy. Because Q13-Q15 were reverse-keyed, we conducted a nonprespecified sensitivity analysis excluding 5 questionnaires with all 15 ratings &#x2265;3 to examine possible acquiescent responding; this criterion did not define invalid responses, and full-sample results remained primary. Filtered means changed minimally for Q1-Q11 (absolute mean difference &#x2264;0.17; <italic>|d|</italic>&#x2264;0.23). In contrast, the 3 perceived cognitive-load-related items (Q13-Q15), reflecting perceived mental effort, perceived information overload, and interface-related distraction, showed lower means, greater dispersion, and larger shifts after filtering (mean difference=&#x2212;0.31 to &#x2212;0.54; <italic>d</italic>=&#x2212;0.27 to &#x2212;0.53), suggesting greater heterogeneity in perceived load and distraction. Overall, the questionnaire results support strong acceptability and perceived educational value of MAPLE-CR; however, perceived cognitive-load-related items remained salient even after filtering. This pattern highlights a broader consideration for AI-supported education: in PBL-like settings, AI may add value less by reducing cognitive demands and more by helping to orchestrate complex discourse and make reasoning visible, while minimizing avoidable interface or coordination demands that can divert effort from learning.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Per-item questionnaire responses (5-point Likert scale) in the full sample (n=26) and filtered sample (n=21)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question</td><td align="left" valign="bottom">Full sample, mean (variance)</td><td align="left" valign="bottom">Filtered, mean (variance)</td><td align="left" valign="bottom">Mean difference</td><td align="left" valign="bottom">Cohen <italic>d</italic></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">Learning satisfaction</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q1. I was satisfied with the overall PBL<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> system experience.</td><td align="left" valign="top">4.19 (0.617)</td><td align="left" valign="top">4.10 (0.658)</td><td align="left" valign="top">&#x2212;0.10</td><td align="left" valign="top">&#x2212;0.12</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q2. The recommended cases matched my learning needs.</td><td align="left" valign="top">4.19 (0.386)</td><td align="left" valign="top">4.14 (0.408)</td><td align="left" valign="top">&#x2212;0.05</td><td align="left" valign="top">&#x2212;0.08</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q3. The system helped me learn the clinical reasoning for this case.</td><td align="left" valign="top">4.27 (0.581)</td><td align="left" valign="top">4.10 (0.562)</td><td align="left" valign="top">&#x2212;0.17</td><td align="left" valign="top">&#x2212;0.23</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q4. The test questions effectively assessed my knowledge.</td><td align="left" valign="top">4.19 (0.386)</td><td align="left" valign="top">4.10 (0.372)</td><td align="left" valign="top">&#x2212;0.10</td><td align="left" valign="top">&#x2212;0.16</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q5. The feedback was accurate and helpful.</td><td align="left" valign="top">4.62 (0.390)</td><td align="left" valign="top">4.57 (0.435)</td><td align="left" valign="top">&#x2212;0.04</td><td align="left" valign="top">&#x2212;0.07</td></tr><tr><td align="left" valign="top" colspan="5">Sense of involvement</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q6. I stayed engaged and actively participated in reasoning/decisions.</td><td align="left" valign="top">4.19 (0.617)</td><td align="left" valign="top">4.10 (0.658)</td><td align="left" valign="top">&#x2212;0.10</td><td align="left" valign="top">&#x2212;0.12</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q7. I could express my views and lead my own reasoning process.</td><td align="left" valign="top">4.50 (0.481)</td><td align="left" valign="top">4.38 (0.522)</td><td align="left" valign="top">&#x2212;0.12</td><td align="left" valign="top">&#x2212;0.17</td></tr><tr><td align="left" valign="top" colspan="5">Perceived support</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q8. Virtual students prompted deeper thinking and participation.</td><td align="left" valign="top">4.38 (0.544)</td><td align="left" valign="top">4.33 (0.508)</td><td align="left" valign="top">&#x2212;0.05</td><td align="left" valign="top">&#x2212;0.07</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q9. The tutor supported me and kept the discussion on track.</td><td align="left" valign="top">4.31 (0.444)</td><td align="left" valign="top">4.19 (0.440)</td><td align="left" valign="top">&#x2212;0.12</td><td align="left" valign="top">&#x2212;0.18</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q10. The system provided timely prompts/corrections when needed.</td><td align="left" valign="top">4.35 (0.380)</td><td align="left" valign="top">4.29 (0.299)</td><td align="left" valign="top">&#x2212;0.06</td><td align="left" valign="top">&#x2212;0.10</td></tr><tr><td align="left" valign="top" colspan="5">Self-efficacy</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q11. I felt more confident analyzing the case and justifying a diagnosis.</td><td align="left" valign="top">4.35 (0.457)</td><td align="left" valign="top">4.29 (0.395)</td><td align="left" valign="top">&#x2212;0.06</td><td align="left" valign="top">&#x2212;0.09</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q12. I can apply the learned reasoning strategies to similar cases.</td><td align="left" valign="top">4.50 (0.404)</td><td align="left" valign="top">4.52 (0.345)</td><td align="left" valign="top">+0.02</td><td align="left" valign="top">+0.04</td></tr><tr><td align="left" valign="top" colspan="5">Cognitive load</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q13. Completing my tasks required substantial mental effort.</td><td align="left" valign="top">3.69 (1.444)</td><td align="left" valign="top">3.38 (1.283)</td><td align="left" valign="top">&#x2212;0.31</td><td align="left" valign="top">&#x2212;0.27</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q14. I often felt information overload during the discussion.</td><td align="left" valign="top">3.15 (2.053)</td><td align="left" valign="top">2.76 (1.705)</td><td align="left" valign="top">&#x2212;0.39</td><td align="left" valign="top">&#x2212;0.28</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Q15. The system prompts or interface often distracted my attention.</td><td align="left" valign="top">2.54 (1.556)</td><td align="left" valign="top">2.00 (0.381)</td><td align="left" valign="top">&#x2212;0.54</td><td align="left" valign="top">&#x2212;0.53</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Full-sample results constitute the main descriptive questionnaire analysis. Filtered results comprise the nonprespecified sensitivity analysis and exclude 5 questionnaires with no ratings below the scale midpoint (all 15 items &#x2265;3); mean differences and Cohen <italic>d</italic> are computed as filtered minus full sample.</p></fn><fn id="table2fn2"><p><sup>b</sup>PBL: problem-based learning.</p></fn></table-wrap-foot></table-wrap><p>Among the 26 MAPLE-CR intervention sessions, transcript-based formative assessment was summarized as supplementary process-oriented evidence aligned with the scaffolding framework across the 5 PBL-relevant dimensions (Figure S1 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). KA and CR had the highest mean scores (4.4 and 3.7) and relatively high minima (3.5 and 3.0), suggesting that most learners articulated coherent and largely correct diagnostic reasoning. By contrast, interaction-oriented dimensions showed greater variability: CC and SR had lower minima (both 1.5), and AP had the lowest central tendency (mean 2.1; minimum 0.5). Together, these findings suggest that even when learners reason accurately, they may not consistently externalize that reasoning through structured discussion, synthesis, and proactive engagement&#x2014;core behaviors in PBL. Notably, the CC dimension reflects learner responsiveness within AI-mediated tutor/peer exchanges rather than real peer-to-peer teamwork.</p><p>In addition, we examined convergent validity by comparing the automated formative scores with independent expert ratings across 26 sessions (3 raters per session; 0&#x2010;5 scale; Table S1 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). The expert panel consisted of 3 attending physicians involved in postgraduate medical education, each with at least 10 years of clinical teaching experience and prior experience in PBL facilitation or CR assessment. Expert raters assessed deidentified session transcripts independently and were blinded to each other&#x2019;s ratings; automated formative scores were not used during expert rating. For the 4 evaluator-scored dimensions, the human experts used the same construct definitions, evidence rules, and 0 to 5 scoring anchors as the evaluator agent; the complete materials are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Interrater reliability of the mean expert ratings was moderate to good across dimensions (ICC[A,3]=0.729&#x2010;0.851): AP 0.851, CR 0.766, KA 0.729, CC 0.807, and SR 0.796. We quantified agreement using Spearman rank correlations (&#x03C1;) between automated scores and the mean expert rating per session, and absolute agreement using MAE. Agreement analyses showed strong rank-order consistency for AP (&#x03C1;=0.88, <italic>P</italic>&#x003C;.001; MAE=0.40), SR (&#x03C1;=0.93, <italic>P</italic>&#x003C;.001; MAE=0.50), and CR (&#x03C1;=0.78, <italic>P</italic>=.007; MAE=0.45). KA demonstrated moderately high agreement (&#x03C1;=0.81, <italic>P</italic>=.005; MAE=0.47). CC exhibited comparatively weaker, though still positive, agreement (&#x03C1;=0.71, <italic>P</italic>=.02) and the largest absolute deviation (MAE=0.68). Overall, these results provide preliminary evidence of convergence between the automated assessment and expert judgments in this small, single-study sample, particularly for participation, reflection, and reasoning. Because the CC reference standard showed good reliability (ICC[A,3]=0.807), the comparatively weaker automated-expert agreement for this dimension suggests that communication-related behaviors may benefit from further rubric calibration and model refinement.</p><p>Figure S2 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> presents an exploratory Spearman rank correlation analysis across session-level evaluation measures within the intervention group. Pairwise complete observations were used: correlations involving questionnaire composites used 21 matched sessions, whereas correlations among test scores and formative metrics used 26 sessions. The 3 perceived cognitive-load-related items (Q13-Q15) were summed as an exploratory cognitive load (COG) composite, and the remaining questionnaire items (Q1-Q12) were summed as an overall experience (EXP) composite. Baseline performance was strongly negatively associated with score gain (pretest score vs score gain, &#x03C1;=&#x2212;0.86), consistent primarily with ceiling effects and regression to the mean. Several transcript-based formative dimensions were positively interrelated, especially AP-SR (<italic>&#x03C1;</italic>=0.77), CR-SR (<italic>&#x03C1;</italic>=0.88), and CC-SR (<italic>&#x03C1;</italic>=0.81), indicating that interactional, reasoning, reflective, and regulatory features tended to co-occur within MAPLE-CR sessions. In contrast, the questionnaire composites showed weak associations with formative metrics and knowledge outcomes (eg, COG-KA, &#x03C1;=&#x2212;0.07; EXP-posttest, &#x03C1;=0.19), suggesting that perceived cognitive load and overall experience should be interpreted as subjective learner-experience indicators rather than direct proxies for performance or process quality. Because no multiplicity adjustment was applied, these correlations should be interpreted descriptively rather than as confirmatory evidence.</p></sec><sec id="s3-2"><title>Voluntary Follow-Up: Descriptive Repeated-Use Patterns</title><p>We descriptively examined repeated-use patterns in within-session test score gain, CR, AP, and the overall evaluation sum across 5 attempts in the voluntary follow-up sample (<xref ref-type="fig" rid="figure5">Figure 5</xref>; n=18). Results for KA, CC, and SR are reported in Figure S3 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Descriptive repeated-use patterns across 5 Multiagent Problem-based Learning Environment for Clinical Reasoning (MAPLE-CR) attempts in the voluntary follow-up sample. Panels show (A) test score gain, (B) overall formative evaluation sum, (C) active participation, and (D) clinical reasoning. Box plots summarize the distribution at each attempt, with the center line indicating the median and the box spanning the 25th to 75th percentiles; the red line connects the median values across attempts to aid visual interpretation of the follow-up pattern.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e95039_fig05.png"/></fig><p>Within-session pretest-to-posttest score differences were positive across all 5 attempts in the voluntary follow-up sample. The median within-session score differences were 15.00 (IQR 6.60-23.30; bootstrap 95% CI 8.30-21.65), 20.00 (IQR 13.30-33.30; bootstrap 95% CI 13.30-26.65), 20.00 (IQR 6.70-20.00; bootstrap 95% CI 6.90-20.00), 16.70 (IQR 9.18-28.35; bootstrap 95% CI 10.00-26.70), and 13.40 (IQR 7.60-25.20; bootstrap 95% CI 8.95-23.50) across attempts 1 through 5, respectively. All CIs were above zero, indicating positive within-session score differences at each attempt.</p><p>For formative outcomes, CR increased markedly from attempt 1 to attempt 2 (median 3.50 to 4.50; Hodges-Lehmann median difference=1.00, 95% CI 1.00&#x2010;1.50; Wilcoxon <italic>P</italic>&#x003C;.001; Cohen <italic>d</italic>=1.739), indicating a higher transcript-based CR score at attempt 2 than at attempt 1 in this descriptive follow-up analysis. In contrast, AP showed a smaller, nonsignificant change (median 1.50 to 2.50; Hodges-Lehmann median difference=0.19, 95% CI 0.00&#x2010;1.00; Wilcoxon <italic>P</italic>=.15; <italic>d</italic>=0.411), indicating a smaller and more heterogeneous observed change in engagement. The overall evaluation sum also increased from attempt 1 to attempt 2 (median 16.00 to 20.00; Hodges-Lehmann median difference=3.50, 95% CI 2.50&#x2010;5.50; Wilcoxon <italic>P</italic>&#x003C;.001; <italic>d</italic>=1.292), indicating a higher aggregate formative score at attempt 2 than at attempt 1.</p><p>From attempt 2 onward, no statistically significant within-subject trends were detected for CR (Friedman &#x03C7;&#x00B2;<sub>3</sub>=6.484<italic>, P</italic>=.09), AP (Friedm<italic>a</italic>n &#x03C7;&#x00B2;<sub>3</sub>=2.088, <italic>P</italic>=.55), or the overall <italic>s</italic>um (Friedman &#x03C7;&#x00B2;<sub>3</sub>=3.034, <italic>P</italic>=.39). These nonsignificant results should not be interpreted as evidence of stabilization; rather, they indicate that no statistically significant later trend was detected in this small voluntary follow-up sample. A sample size sensitivity calculation indicated that detecting a smaller but educationally meaningful within-participant change of <italic>d</italic>=0.50 with 2-sided &#x03B1;=.05 and 80% power would require approximately 34 completed follow-up participants, exceeding the available follow-up sample (n=18).</p><p>Additionally, student case-study pathways are visualized in Figure S4 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>. Although all learners began with the same thoracic and airway surgery case, half diversified into other systems in the second attempt. By the fifth attempt, students had collectively explored over 9 distinct medical domains, with some choosing to revisit similar systems (eg, hepatobiliary or cardiovascular) and others branching into entirely new areas (eg, pediatrics or oncology). These patterns suggest heterogeneity in follow-up case selection; whether this reflected learner interests, perceived needs, or recommendation effects could not be determined from the present data.</p><p>We also recorded the time learners spent on each PBL discussion, excluding testing and assessment (Figure S5 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). The overall median was 67.39 (IQR 54.7-90.0) minutes. The median discussion duration decreased from 70.87 (IQR 58.5-93.2) minutes at attempt 1 to 54.87 (IQR 48.1-82.9) minutes at attempt 5, a 22.6% reduction. Despite session-to-session fluctuations, discussions were shorter by the fifth attempt; however, this pattern may reflect increasing familiarity, case differences, or reduced engagement.</p></sec><sec id="s3-3"><title>Open-Ended Questionnaire</title><p>To contextualize learners&#x2019; perceived strengths and limitations of MAPLE-CR, we descriptively summarized the open-ended feedback entries and grouped them into nonmutually exclusive categories. The complete categorized comments are provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p><p>Positive feedback most often described structured CR support (5 feedback entries), perceived learning gains (4 entries), reduced pressure and freer expression compared with classroom PBL (4 entries), and content coverage or case realism (3 entries). These comments suggested that learners perceived the simulated PBL format as immersive, flexible, and helpful for organizing diagnostic and management reasoning.</p><p>Areas for improvement included interaction depth, flow, or tutor questioning (3 feedback entries), assessment or questionnaire design (2 entries), content repetition or concision (2 entries), and interface usability or navigation (2 entries). Single feedback entries also noted conversational accuracy or AI misunderstanding, time or session-control burden, and requests for additional learning-tool support such as discussion filtering or mind-map functions.</p><p>Overall, these open-ended responses provided descriptive support for perceived usefulness and learner acceptability while identifying concrete areas for refinement in conversational accuracy, interaction design, assessment design, and workload.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This exploratory randomized educational evaluation provides preliminary evidence that MAPLE-CR is a feasible and educationally valuable AI-supported PBL environment for CR training. In the randomized component, learners using MAPLE-CR achieved higher posttest scores and greater pretest-posttest gains than those in the self-study control, suggesting a large short-term difference under the present study conditions rather than definitive efficacy evidence. Learner-reported outcomes indicated high acceptability and perceived educational value across satisfaction, involvement, perceived support, and self-efficacy. Because engagement with generative AI may vary with learner characteristics and implementation context, these acceptability findings should be interpreted within the present population and setting [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Notably, however, learners also reported relatively high ratings on cognitive-load items. This pattern should not be interpreted as evidence that reducing real peer coordination is educationally beneficial or that AI-mediated PBL can replace human group interaction. Rather, for MAPLE-CR as a supplementary PBL learning environment, the design challenge is to preserve productive reasoning and dialogic engagement while minimizing avoidable interface or workflow demands. From a cognitive load perspective, this means preserving productive challenge associated with reasoning and explanation [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>] while reducing extraneous load introduced by the learning interface and workflow structure [<xref ref-type="bibr" rid="ref34">34</xref>]. From a social constructivist perspective, these findings are consistent with the possibility that AI-driven scaffolding can function as a provisional bridge within the learner&#x2019;s Zone of Proximal Development [<xref ref-type="bibr" rid="ref35">35</xref>]. One interpretation is that MAPLE-CR&#x2019;s orchestration of the PBL process, including structured summaries, next-step prompts, and feedback loops, may have helped learners direct effort toward higher-level reasoning, although the specific contribution of each scaffolding component requires further investigation.</p><p>The control condition produced a measurable median gain of +6.70 (IQR 0.00-13.40) points, whereas the between-group median difference in gain was 6.60 points (bootstrap 95% CI 1.60-14.95). Using these values as a descriptive anchor, a substantial portion of the short-term pretest-posttest improvement was also observed under structured exposure to the same case materials, while the additional separation represents the incremental difference observed for MAPLE-CR under this comparator. This comparison is descriptive only and should not be interpreted as a causal partitioning of the effect. The present design evaluated MAPLE-CR as an integrated learning package. The observed between-group difference may therefore reflect the combined influence of platform use, orientation, AI-mediated interaction and feedback, additional structured exposure, cognitive and regulatory scaffolding, novelty, and motivational engagement rather than the isolated effect of the AI-PBL scaffolding mechanisms; nor does the design establish noninferiority to faculty-facilitated, peer-driven PBL.</p><p>Across the voluntary 4-week follow-up, descriptive patterns showed higher CR and aggregate formative scores at attempt 2 than at attempt 1, followed by statistically uncertain later trajectories. Given the absence of a control group, self-selected participation, heterogeneous case pathways, and the small sample, these observations do not establish maintenance, stabilization, or sustained learning; they could reflect adaptation, ceiling or novelty effects, effort, case heterogeneity, or limited power [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. They instead support the feasibility of repeated MAPLE-CR use across different cases and identify patterns that warrant evaluation in controlled longitudinal studies, including studies of adaptive prompts or progressively more complex cases [<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>In addition, a key pattern was the dissociation between relatively strong knowledge/reasoning performance and more heterogeneous interactional performance (eg, participation, collaboration, and reflection). This aligns with prior work suggesting that competence in CR does not automatically translate into observable group-process skills, particularly in technology-mediated settings where turn-taking norms and role clarity are less salient than in facilitator-led discussions [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. Accordingly, AI-enabled PBL designs may need to scaffold not only what learners think (eg, prompts for hypothesis generation and justification) but also how learners participate and co-construct understanding (eg, turn-taking, accountability structures, and synthesis routines). Within the present study, MAPLE-CR represents an empirically evaluated implementation of conversational AI that spans multiple phases of the PBL cycle rather than serving as an occasional question-answering aid. The findings also indicate where additional process support is needed.</p></sec><sec id="s4-2"><title>Implications for Medical Education</title><p>Beyond our local internship context, a key design implication of MAPLE-CR is a set of potentially transferable scaffolding features for supporting scalable CR practice: embedding cognitive prompts that require explanation and revision, social structures that elicit multivoice challenge rather than simple learner-AI dyads, and regulatory supports that make phases and feedback loops explicit. When trained facilitators, standardized patients, or stable small-group scheduling are difficult to sustain, such a pattern may offer a scalable supplement for presession rehearsal, between-session deliberate practice, or postsession consolidation within blended curricula [<xref ref-type="bibr" rid="ref39">39</xref>]. Because simulated group-process behaviors varied across sessions, AI is unlikely to fully reproduce the nuance of human facilitation [<xref ref-type="bibr" rid="ref23">23</xref>]; implementation may therefore benefit from brief learner orientation (expectations for participation, justification, and turn-taking) and a short postsession debrief to normalize reflection and calibrate engagement. In future implementations, transcript summaries and formative analytics could be evaluated as tools to help faculty identify sessions with recurrent needs (eg, persistently low collaboration or superficial justification) and provide focused coaching without routine full-transcript review. At its current maturity, MAPLE-CR is best positioned as a scalable, formative, low-stakes supplement for structured PBL-like practice rather than as a replacement for faculty-facilitated PBL; further multisite evidence is needed before considering high-stakes or summative uses, particularly across languages, curricula, and assessment regimes.</p></sec><sec id="s4-3"><title>Safety</title><p>MAPLE-CR is an educational tool rather than a clinical decision support system and is intended for formative, low-stakes learning rather than summative decision-making. Tutor and evaluator functions are bounded by case objectives and curated learning resources, and feedback emphasizes reasoning processes (eg, justification, revision, reflection) rather than prescriptive management recommendations. Cases and prompts are designed for training purposes and do not involve identifiable patient information; discussion transcripts and learning records are stored and analyzed in deidentified form for educational feedback and research. Although no safety incidents were observed in this study, future iterations will strengthen safeguards to mitigate misinformation and over-reliance and will maintain ongoing monitoring and a clear reporting pathway for any concerning outputs [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref41">41</xref>].</p></sec><sec id="s4-4"><title>Limitations and Suggestions for Future Research</title><p>This study has limitations. First, no formal a priori sample size calculation was conducted. Although the sensitivity analysis showed that the observed effect magnitude exceeded the approximate large-effect threshold detectable with the available sample, this descriptive analysis does not establish prospective power sufficiency; the study was not prospectively powered as a confirmatory efficacy trial, and the precision of the estimated effect remains limited. The findings should therefore be confirmed in larger, prospectively powered, multicenter trials with stronger active comparators. The adapted postsession questionnaire was not validated specifically for AI-supported PBL, so learner-experience findings should be interpreted descriptively. Second, the trial was retrospectively registered, which limits full compliance with prospective RCT registration expectations and should be considered when interpreting the randomized component. Third, the control condition was self-study, which represents a relatively modest active comparator; this comparator did not isolate the contribution of specific PBL scaffolding mechanisms from other features of MAPLE-CR, such as AI-mediated dialogue, simulated peer/tutor interaction, novelty, or motivational engagement; future studies should benchmark MAPLE-CR against stronger comparators that more closely reflect established PBL pedagogy (eg, facilitator-led small-group PBL), near-peer facilitation, or other structured instructional modalities [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. The conditions were also not matched for total structured exposure because only the MAPLE-CR group received the 40-minute orientation. The randomized component also used a single standardized lung cancer/thoracic-airway surgery case to ensure comparable exposure across groups; therefore, the observed effect may not generalize across clinical domains, case complexity levels, or scenario types. The voluntary follow-up supported repeated use across additional cases but was not designed to determine whether MAPLE-CR effects differ by case type or complexity. In addition, stochastic LLM generation may have introduced variability in the tutor and peer responses that learners received, creating a potential source of within-condition interference that was not explicitly modeled. Fourth, outcomes emphasized immediate knowledge performance and learner experience; although the follow-up suggested feasibility for repeated cross-case use, participation was voluntary and may overestimate engagement and perceived benefit. The follow-up included only 18 participants and was not powered for confirmatory inference about longer-term learning trajectories. Fifth, we did not include standardized external measures of retention and transfer (eg, objective structured clinical examination-style assessments) [<xref ref-type="bibr" rid="ref43">43</xref>]. Although the transcript-derived formative dimensions were aligned with the proposed scaffolding framework and showed preliminary convergence with expert ratings, they were not prospectively designed to test mediation or isolate specific cognitive, social-interactional, or regulatory mechanisms. Future studies should incorporate prespecified process measures to examine how specific scaffolding functions contribute to learning outcomes.</p><p>Beyond these study-level limitations, future research should move from asking whether AI-supported PBL &#x201C;works&#x201D; to testing how and for whom it works. Mechanism-focused studies could examine whether improvements are mediated by process indicators aligned to the proposed cognitive, social-interactional, and regulatory functions (eg, quality of justification and revision, participation patterns and responsiveness, adherence to PBL phases, and uptake of feedback across cases). Work is also needed to identify boundary conditions and learner-level moderators (eg, baseline knowledge, self-efficacy, language/communication confidence, and AI literacy), given the observed heterogeneity in group-process behaviors. Finally, the higher scores observed at attempt 2 and statistically uncertain later trajectories suggest a need to investigate adaptive progression and scaffold fading&#x2014;such as progressively more complex cases, calibrated prompts, or feedback designs&#x2014;under controlled longitudinal conditions [<xref ref-type="bibr" rid="ref24">24</xref>]. Multisite and real-world implementation studies should evaluate feasibility, faculty workload, and sustainability alongside learning outcomes, and clarify appropriate use as formative support rather than high-stakes assessment.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This study provides preliminary evidence that MAPLE-CR is a viable AI-supported scaffold for PBL and suggests that generative AI can serve a pedagogical role beyond functioning as a mere information source. MAPLE-CR was associated with larger short-term knowledge gains than self-study and showed high acceptability and feasible repeated use, but the design did not isolate specific scaffolding mechanisms. At the same time, the finding that gains in knowledge and CR may emerge earlier than improvements in observable participation and collaborative process skills highlights the complexity of AI-mediated instructional design. As a scalable supplement, MAPLE-CR may provide additional opportunities for structured PBL-like practice when faculty facilitation and repeated small-group sessions are difficult to sustain. These findings require confirmation in larger, prospectively powered, multicenter trials with stronger active comparators and longer-term measures of retention, transfer, and cross-case generalizability.</p></sec></sec></body><back><ack><p>Generative AI (ChatGPT, GPT-5.6 Sol, OpenAI) was used solely to improve the language and readability of the manuscript. All AI-assisted edits were reviewed by the authors, who take full responsibility for the final content. We also thank all medical interns who participated in the study for their time, effort, and valuable engagement.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the National Natural Science Foundation of China (grant 62577049).</p></sec><sec><title>Data Availability</title><p>The deidentified data supporting the reported results are available from the corresponding author on reasonable request. The multiagent problem-based learning environment for clinical reasoning code is available in a public GitHub repository [<xref ref-type="bibr" rid="ref44">44</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>ZZ led system development and data processing. BW and ST led data collection. ZZ and BW drafted the initial manuscript. ST, HH, and PH supervised the study, reviewed the manuscript, and contributed substantive intellectual revision. All authors contributed to the study design and interpretation of findings, approved the final manuscript, and agreed to be accountable for all aspects of the work.</p></fn><fn fn-type="conflict"><p>ZZ led the development of MAPLE-CR, the AI-supported PBL platform evaluated in this study; this role is disclosed for transparency. All authors declare no competing interests.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AP</term><def><p>active participation</p></def></def-item><def-item><term id="abb2">CC</term><def><p>communication and collaboration</p></def></def-item><def-item><term id="abb3">CONSORT-eHEALTH</term><def><p>Consolidated Standards of Reporting Trials of Electronic and Mobile Health Applications and Online Telehealth</p></def></def-item><def-item><term id="abb4">CR</term><def><p>clinical reasoning</p></def></def-item><def-item><term id="abb5">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb6">KA</term><def><p>knowledge accuracy</p></def></def-item><def-item><term id="abb7">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb8">MAE</term><def><p>mean absolute error</p></def></def-item><def-item><term id="abb9">MAPLE-CR</term><def><p>Multiagent Problem-based Learning Environment for Clinical Reasoning</p></def></def-item><def-item><term id="abb10">PBL</term><def><p>problem-based learning</p></def></def-item><def-item><term id="abb11">SR</term><def><p>summarization and reflection</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hmelo-Silver</surname><given-names>CE</given-names> </name></person-group><article-title>Problem-based learning: what and how do students learn?</article-title><source>Educ Psychol Rev</source><year>2004</year><month>09</month><volume>16</volume><issue>3</issue><fpage>235</fpage><lpage>266</lpage><pub-id pub-id-type="doi">10.1023/B:EDPR.0000034022.16470.f3</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dineen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lazarus</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Stephens</surname><given-names>GC</given-names> </name></person-group><article-title>Uncertainty experienced by newly qualified doctors during the transition to internship</article-title><source>Med Educ</source><year>2025</year><month>10</month><volume>59</volume><issue>10</issue><fpage>1079</fpage><lpage>1093</lpage><pub-id pub-id-type="doi">10.1111/medu.15692</pub-id><pub-id pub-id-type="medline">40156179</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Su</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>L</given-names> </name></person-group><article-title>Application of problem based learning (PBL) and case based learning (CBL) in the teaching of international classification of diseases encoding</article-title><source>Sci Rep</source><year>2023</year><month>09</month><day>14</day><volume>13</volume><issue>1</issue><fpage>15220</fpage><pub-id pub-id-type="doi">10.1038/s41598-023-42175-1</pub-id><pub-id pub-id-type="medline">37709817</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dochy</surname><given-names>F</given-names> </name><name name-style="western"><surname>Segers</surname><given-names>M</given-names> </name><name name-style="western"><surname>Van den Bossche</surname><given-names>P</given-names> </name><name name-style="western"><surname>Gijbels</surname><given-names>D</given-names> </name></person-group><article-title>Effects of problem-based learning: a meta-analysis</article-title><source>Learn Instr</source><year>2003</year><month>10</month><volume>13</volume><issue>5</issue><fpage>533</fpage><lpage>568</lpage><pub-id pub-id-type="doi">10.1016/S0959-4752(02)00025-7</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidt</surname><given-names>HG</given-names> </name><name name-style="western"><surname>Rotgans</surname><given-names>JI</given-names> </name><name name-style="western"><surname>Yew</surname><given-names>EHJ</given-names> </name></person-group><article-title>The process of problem-based learning: what works and why</article-title><source>Med Educ</source><year>2011</year><month>08</month><volume>45</volume><issue>8</issue><fpage>792</fpage><lpage>806</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2923.2011.04035.x</pub-id><pub-id pub-id-type="medline">21752076</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dolmans</surname><given-names>DHJM</given-names> </name><name name-style="western"><surname>Gijselaers</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Moust</surname><given-names>JHC</given-names> </name><name name-style="western"><surname>de Grave</surname><given-names>WS</given-names> </name><name name-style="western"><surname>Wolfhagen</surname><given-names>IHAP</given-names> </name><name name-style="western"><surname>van der Vleuten</surname><given-names>CPM</given-names> </name></person-group><article-title>Trends in research on the tutor in problem-based learning: conclusions and implications for educational practice and research</article-title><source>Med Teach</source><year>2002</year><month>03</month><volume>24</volume><issue>2</issue><fpage>173</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1080/01421590220125277</pub-id><pub-id pub-id-type="medline">12098437</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jones</surname><given-names>NL</given-names> </name><name name-style="western"><surname>Peiffer</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Lambros</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Developing a problem-based learning (PBL) curriculum for professionalism and scientific integrity training for biomedical graduate students</article-title><source>J Med Ethics</source><year>2010</year><month>10</month><volume>36</volume><issue>10</issue><fpage>614</fpage><lpage>619</lpage><pub-id pub-id-type="doi">10.1136/jme.2009.035220</pub-id><pub-id pub-id-type="medline">20797979</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Monrouxe</surname><given-names>LV</given-names> </name><name name-style="western"><surname>Rees</surname><given-names>CE</given-names> </name></person-group><article-title>The socialisation of mistreatment in the healthcare workplace: moving beyond narrative content to analyse educator data as discourse</article-title><source>Med Educ</source><year>2023</year><month>10</month><volume>57</volume><issue>10</issue><fpage>882</fpage><lpage>885</lpage><pub-id pub-id-type="doi">10.1111/medu.15122</pub-id><pub-id pub-id-type="medline">37183307</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Farquhar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kossen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kuhn</surname><given-names>L</given-names> </name><name name-style="western"><surname>Gal</surname><given-names>Y</given-names> </name></person-group><article-title>Detecting hallucinations in large language models using semantic entropy</article-title><source>Nature</source><year>2024</year><month>06</month><volume>630</volume><issue>8017</issue><fpage>625</fpage><lpage>630</lpage><pub-id pub-id-type="doi">10.1038/s41586-024-07421-0</pub-id><pub-id pub-id-type="medline">38898292</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thesen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Park</surname><given-names>SH</given-names> </name></person-group><article-title>A generative AI teaching assistant for personalized learning in medical education</article-title><source>NPJ Digit Med</source><year>2025</year><month>11</month><day>4</day><volume>8</volume><issue>1</issue><fpage>627</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02022-1</pub-id><pub-id pub-id-type="medline">41188616</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shalong</surname><given-names>W</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Bin</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Enhancing self-directed learning with custom GPT AI facilitation among medical students: a randomized controlled trial</article-title><source>Med Teach</source><year>2025</year><month>07</month><volume>47</volume><issue>7</issue><fpage>1126</fpage><lpage>1133</lpage><pub-id pub-id-type="doi">10.1080/0142159X.2024.2413023</pub-id><pub-id pub-id-type="medline">39425996</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nguyen</surname><given-names>NP</given-names> </name><name name-style="western"><surname>Tran</surname><given-names>P</given-names> </name></person-group><article-title>Bridging the mentorship divide: how large language models could reshape medical workforce equity</article-title><source>NPJ Digit Med</source><year>2026</year><month>01</month><day>9</day><volume>9</volume><issue>1</issue><fpage>29</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02167-z</pub-id><pub-id pub-id-type="medline">41513946</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Holderried</surname><given-names>F</given-names> </name><name name-style="western"><surname>Stegemann-Philipps</surname><given-names>C</given-names> </name><name name-style="western"><surname>Herrmann-Werner</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A language model-powered simulated patient with automated feedback for history taking: prospective study</article-title><source>JMIR Med Educ</source><year>2024</year><month>08</month><day>16</day><volume>10</volume><issue>1</issue><fpage>e59213</fpage><pub-id pub-id-type="doi">10.2196/59213</pub-id><pub-id pub-id-type="medline">39150749</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Development and evaluation of a retrieval-augmented large language model framework for enhancing endodontic education</article-title><source>Int J Med Inform</source><year>2025</year><month>11</month><volume>203</volume><fpage>106006</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.106006</pub-id><pub-id pub-id-type="medline">40479778</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>F</given-names> </name></person-group><article-title>Reconfiguring tutor-peer roles to foster critical thinking in medical PBL assisted by tutor-managed ChatGPT</article-title><source>Med Educ</source><year>2026</year><month>08</month><volume>60</volume><issue>8</issue><fpage>894</fpage><lpage>907</lpage><pub-id pub-id-type="doi">10.1111/medu.70100</pub-id><pub-id pub-id-type="medline">41308655</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hui</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zewu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Jiao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>C</given-names> </name></person-group><article-title>Application of ChatGPT-assisted problem-based learning teaching method in clinical medical education</article-title><source>BMC Med Educ</source><year>2025</year><month>01</month><day>11</day><volume>25</volume><issue>1</issue><fpage>50</fpage><pub-id pub-id-type="doi">10.1186/s12909-024-06321-1</pub-id><pub-id pub-id-type="medline">39799356</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barrows</surname><given-names>HS</given-names> </name></person-group><article-title>Problem&#x2010;based learning in medicine and beyond: a brief overview</article-title><source>New Dir Teach Learn</source><year>1996</year><month>12</month><volume>1996</volume><issue>68</issue><fpage>3</fpage><lpage>12</lpage><pub-id pub-id-type="doi">10.1002/tl.37219966804</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Basil</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>W</given-names> </name><name name-style="western"><surname>Hajeomar</surname><given-names>R</given-names> </name><name name-style="western"><surname>Strawbridge</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lynch</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mukhalalati</surname><given-names>B</given-names> </name></person-group><article-title>A scoping review of the use of generative artificial intelligence tools in health profession education</article-title><source>BMC Med Educ</source><year>2026</year><month>01</month><day>23</day><volume>26</volume><issue>1</issue><fpage>291</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-08527-3</pub-id><pub-id pub-id-type="medline">41578287</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pangastuti</surname><given-names>D</given-names> </name><name name-style="western"><surname>Widiasih</surname><given-names>N</given-names> </name><name name-style="western"><surname>Soemantri</surname><given-names>D</given-names> </name></person-group><article-title>Piloting a constructive feedback model for problem-based learning in medical education</article-title><source>Korean J Med Educ</source><year>2022</year><month>06</month><volume>34</volume><issue>2</issue><fpage>131</fpage><lpage>143</lpage><pub-id pub-id-type="doi">10.3946/kjme.2022.225</pub-id><pub-id pub-id-type="medline">35676880</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Izquierdo-Condoy</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Arias-Intriago</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tello-De-la-Torre</surname><given-names>A</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ortiz-Prado</surname><given-names>E</given-names> </name></person-group><article-title>Generative artificial intelligence in medical education: enhancing critical thinking or undermining cognitive autonomy?</article-title><source>J Med Internet Res</source><year>2025</year><month>11</month><day>3</day><volume>27</volume><fpage>e76340</fpage><pub-id pub-id-type="doi">10.2196/76340</pub-id><pub-id pub-id-type="medline">41183320</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vogel</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wecker</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kollar</surname><given-names>I</given-names> </name><name name-style="western"><surname>Fischer</surname><given-names>F</given-names> </name></person-group><article-title>Socio-cognitive scaffolding with computer-supported collaboration scripts: a meta-analysis</article-title><source>Educ Psychol Rev</source><year>2017</year><month>09</month><volume>29</volume><issue>3</issue><fpage>477</fpage><lpage>511</lpage><pub-id pub-id-type="doi">10.1007/s10648-016-9361-7</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kreijns</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kirschner</surname><given-names>PA</given-names> </name><name name-style="western"><surname>Jochems</surname><given-names>W</given-names> </name></person-group><article-title>Identifying the pitfalls for social interaction in computer-supported collaborative learning environments: a review of the research</article-title><source>Comput Human Behav</source><year>2003</year><month>05</month><volume>19</volume><issue>3</issue><fpage>335</fpage><lpage>353</lpage><pub-id pub-id-type="doi">10.1016/S0747-5632(02)00057-2</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kassab</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Hamdy</surname><given-names>H</given-names> </name><name name-style="western"><surname>Mamede</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schmidt</surname><given-names>H</given-names> </name></person-group><article-title>Influence of tutor interventions and group process on medical students&#x2019; engagement in problem-based learning</article-title><source>Med Educ</source><year>2024</year><month>11</month><volume>58</volume><issue>11</issue><fpage>1315</fpage><lpage>1323</lpage><pub-id pub-id-type="doi">10.1111/medu.15387</pub-id><pub-id pub-id-type="medline">38563548</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Belland</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hannafin</surname><given-names>MJ</given-names> </name></person-group><article-title>A framework for designing scaffolds that improve motivation and cognition</article-title><source>Educ Psychol</source><year>2013</year><month>10</month><volume>48</volume><issue>4</issue><fpage>243</fpage><lpage>270</lpage><pub-id pub-id-type="doi">10.1080/00461520.2013.838920</pub-id><pub-id pub-id-type="medline">24273351</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>D</given-names> </name><etal/></person-group><article-title>ReAct: synergizing reasoning and acting in language models</article-title><access-date>2026-08-18</access-date><conf-name>11th International Conference on Learning Representations, ICLR 2023</conf-name><conf-date>May 1-5, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://collaborate.princeton.edu/en/publications/react-synergizing-reasoning-and-acting-in-language-models/">https://collaborate.princeton.edu/en/publications/react-synergizing-reasoning-and-acting-in-language-models/</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Bansal</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>AutoGen: enabling next-gen LLM applications via multi-agent conversation</article-title><access-date>2026-08-18</access-date><conf-name>First Conference on Language Modeling (COLM 2024)</conf-name><conf-date>Oct 7-9, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.microsoft.com/en-us/research/publication/autogen-enabling-next-gen-llm-applications-via-multi-agent-conversation-framework/">https://www.microsoft.com/en-us/research/publication/autogen-enabling-next-gen-llm-applications-via-multi-agent-conversation-framework/</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><article-title>Alibaba cloud model studio</article-title><source>Alibaba Cloud</source><access-date>2026-07-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.alibabacloud.com/help/en/model-studio/">https://www.alibabacloud.com/help/en/model-studio/</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ekelund</surname><given-names>K</given-names> </name><name name-style="western"><surname>O&#x2019;Regan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Dieckmann</surname><given-names>P</given-names> </name><name name-style="western"><surname>&#x00D8;stergaard</surname><given-names>D</given-names> </name><name name-style="western"><surname>Watterson</surname><given-names>L</given-names> </name></person-group><article-title>Evaluation of the simulation based training quality assurance tool (SBT-QA10) as a measure of learners&#x2019; perceptions during the action phase of simulation</article-title><source>BMC Med Educ</source><year>2023</year><month>05</month><day>1</day><volume>23</volume><issue>1</issue><fpage>290</fpage><pub-id pub-id-type="doi">10.1186/s12909-023-04273-6</pub-id><pub-id pub-id-type="medline">37127593</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yamamoto</surname><given-names>A</given-names> </name><name name-style="western"><surname>Koda</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ogawa</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Enhancing medical interview skills through AI-simulated patient interactions: nonrandomized controlled trial</article-title><source>JMIR Med Educ</source><year>2024</year><month>09</month><day>23</day><volume>10</volume><issue>1</issue><fpage>e58753</fpage><pub-id pub-id-type="doi">10.2196/58753</pub-id><pub-id pub-id-type="medline">39312284</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arpaci</surname><given-names>I</given-names> </name><name name-style="western"><surname>Ku&#x015F;ci</surname><given-names>I</given-names> </name><name name-style="western"><surname>Gibreel</surname><given-names>O</given-names> </name></person-group><article-title>The role of personality traits in predicting educational use of generative AI in higher education</article-title><source>Sci Rep</source><year>2025</year><month>08</month><day>19</day><volume>15</volume><issue>1</issue><fpage>30440</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-16339-0</pub-id><pub-id pub-id-type="medline">40830409</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arpaci</surname><given-names>I</given-names> </name><name name-style="western"><surname>Gibreel</surname><given-names>O</given-names> </name><name name-style="western"><surname>Al-Sharafi</surname><given-names>MA</given-names> </name></person-group><article-title>A hybrid SEM-ANN approach to predicting generative AI use in higher education through knowledge management</article-title><source>Comput Hum Behav Rep</source><year>2026</year><month>05</month><volume>22</volume><fpage>101044</fpage><pub-id pub-id-type="doi">10.1016/j.chbr.2026.101044</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sweller</surname><given-names>J</given-names> </name><name name-style="western"><surname>van Merrienboer</surname><given-names>JJG</given-names> </name><name name-style="western"><surname>Paas</surname><given-names>FGWC</given-names> </name></person-group><article-title>Cognitive architecture and instructional design</article-title><source>Educ Psychol Rev</source><year>1998</year><month>09</month><volume>10</volume><issue>3</issue><fpage>251</fpage><lpage>296</lpage><pub-id pub-id-type="doi">10.1023/A:1022193728205</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paas</surname><given-names>F</given-names> </name><name name-style="western"><surname>Renkl</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sweller</surname><given-names>J</given-names> </name></person-group><article-title>Cognitive load theory and instructional design: recent developments</article-title><source>Educ Psychol</source><year>2003</year><month>01</month><day>1</day><volume>38</volume><issue>1</issue><fpage>1</fpage><lpage>4</lpage><pub-id pub-id-type="doi">10.1207/S15326985EP3801_1</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Jong</surname><given-names>T</given-names> </name></person-group><article-title>Cognitive load theory, educational research, and instructional design: some food for thought</article-title><source>Instr Sci</source><year>2010</year><month>03</month><volume>38</volume><issue>2</issue><fpage>105</fpage><lpage>134</lpage><pub-id pub-id-type="doi">10.1007/s11251-009-9110-0</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Vygotsky</surname><given-names>LS</given-names> </name><name name-style="western"><surname>Cole</surname><given-names>M</given-names> </name><name name-style="western"><surname>John-Steiner</surname><given-names>V</given-names> </name><name name-style="western"><surname>Scribner</surname><given-names>S</given-names> </name><name name-style="western"><surname>Souberman</surname><given-names>E</given-names> </name></person-group><source>Mind in Society: The Development of Higher Psychological Processes</source><year>1978</year><publisher-name>Harvard University Press</publisher-name><pub-id pub-id-type="doi">10.2307/j.ctvjf9vz4</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barry Issenberg</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mcgaghie</surname><given-names>WC</given-names> </name><name name-style="western"><surname>Petrusa</surname><given-names>ER</given-names> </name><name name-style="western"><surname>Lee Gordon</surname><given-names>D</given-names> </name><name name-style="western"><surname>Scalese</surname><given-names>RJ</given-names> </name></person-group><article-title>Features and uses of high-fidelity medical simulations that lead to effective learning: a BEME systematic review</article-title><source>Med Teach</source><year>2005</year><month>01</month><volume>27</volume><issue>1</issue><fpage>10</fpage><lpage>28</lpage><pub-id pub-id-type="doi">10.1080/01421590500046924</pub-id><pub-id pub-id-type="medline">16147767</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Motola</surname><given-names>I</given-names> </name><name name-style="western"><surname>Devine</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Sullivan</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Issenberg</surname><given-names>SB</given-names> </name></person-group><article-title>Simulation in healthcare education: a best evidence practical guide. AMEE Guide No. 82</article-title><source>Med Teach</source><year>2013</year><month>10</month><volume>35</volume><issue>10</issue><fpage>e1511</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.3109/0142159X.2013.818632</pub-id><pub-id pub-id-type="medline">23941678</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zamecnik</surname><given-names>A</given-names> </name><name name-style="western"><surname>Villa-Torrano</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kovanovi&#x0107;</surname><given-names>V</given-names> </name><etal/></person-group><article-title>The cohesion of small groups in technology-mediated learning environments: a systematic literature review</article-title><source>Educ Res Rev</source><year>2022</year><month>02</month><volume>35</volume><fpage>100427</fpage><pub-id pub-id-type="doi">10.1016/j.edurev.2021.100427</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elendu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Amaechi</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Okatta</surname><given-names>AU</given-names> </name><etal/></person-group><article-title>The impact of simulation-based training in medical education: a review</article-title><source>Medicine (Baltimore)</source><year>2024</year><month>07</month><day>5</day><volume>103</volume><issue>27</issue><fpage>e38813</fpage><pub-id pub-id-type="doi">10.1097/MD.0000000000038813</pub-id><pub-id pub-id-type="medline">38968472</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Quennelle</surname><given-names>S</given-names> </name><name name-style="western"><surname>Malekzadeh-Milani</surname><given-names>S</given-names> </name><name name-style="western"><surname>Garcelon</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Active learning for extracting rare adverse events from electronic health records: a study in pediatric cardiology</article-title><source>Int J Med Inform</source><year>2025</year><month>03</month><volume>195</volume><fpage>105761</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2024.105761</pub-id><pub-id pub-id-type="medline">39689449</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>L&#x00F3;pez-&#x00DA;beda</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mart&#x00ED;n-Noguerol</surname><given-names>T</given-names> </name><name name-style="western"><surname>Luna</surname><given-names>A</given-names> </name></person-group><article-title>Integrating semantic retrieval and chain-of-thought reasoning in small language models for SNOMED CT normalization</article-title><source>Int J Med Inform</source><year>2026</year><month>05</month><volume>211</volume><fpage>106340</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2026.106340</pub-id><pub-id pub-id-type="medline">41678976</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elhilali</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ngo</surname><given-names>ASH</given-names> </name><name name-style="western"><surname>Reichenpfader</surname><given-names>D</given-names> </name><name name-style="western"><surname>Denecke</surname><given-names>K</given-names> </name></person-group><article-title>Large language model-based patient simulation to foster communication skills in health care professionals: user-centered development and usability study</article-title><source>JMIR Med Educ</source><year>2025</year><month>12</month><day>12</day><volume>11</volume><fpage>e81271</fpage><pub-id pub-id-type="doi">10.2196/81271</pub-id><pub-id pub-id-type="medline">41385781</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harden</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Gleeson</surname><given-names>FA</given-names> </name></person-group><article-title>Assessment of clinical competence using an objective structured clinical examination (OSCE)</article-title><source>Med Educ</source><year>1979</year><month>01</month><volume>13</volume><issue>1</issue><fpage>41</fpage><lpage>54</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2923.1979.tb00918.x</pub-id><pub-id pub-id-type="medline">763183</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="web"><article-title>MAPLE-CR</article-title><source>GitHub</source><year>2026</year><access-date>2026-03-09</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/eMoLii/PBL">https://github.com/eMoLii/PBL</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Detailed description of the technical implementation of multiagent problem-based learning environment for clinical reasoning.</p><media xlink:href="mededu_v12i1e95039_app1.docx" xlink:title="DOCX File, 28 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Video demonstration of the MAPLE-CR system workflow and user interface.</p><media xlink:href="mededu_v12i1e95039_app2.mp4" xlink:title="MP4 File, 56462 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Blueprint of the pretest and posttest design.</p><media xlink:href="mededu_v12i1e95039_app3.docx" xlink:title="DOCX File, 32 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Additional experimental and follow-up results.</p><media xlink:href="mededu_v12i1e95039_app4.pdf" xlink:title="PDF File, 3128 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Descriptive categorization of open-ended feedback.</p><media xlink:href="mededu_v12i1e95039_app5.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app6"><label>Checklist 1</label><p>CONSORT-eHEALTH checklist (V 1.6.1).</p><media xlink:href="mededu_v12i1e95039_app6.pdf" xlink:title="PDF File, 2951 KB"/></supplementary-material></app-group></back></article>