<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id><journal-id journal-id-type="publisher-id">mededu</journal-id><journal-id journal-id-type="index">20</journal-id><journal-title>JMIR Medical Education</journal-title><abbrev-journal-title>JMIR Med Educ</abbrev-journal-title><issn pub-type="epub">2369-3762</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v12i1e92964</article-id><article-id pub-id-type="doi">10.2196/92964</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Multiagent Large Language Model Framework for Psychotherapy Fidelity Assessment in Motivational Interviewing and Cognitive Behavioral Therapy Training: Cross-Sectional, Simulation-Based Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Kamaleddin</surname><given-names>Mohammad Amin</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mirjalili</surname><given-names>Mina</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Barzegar</surname><given-names>Reza</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Trung Le</surname><given-names>Nghia</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cote</surname><given-names>Zachary</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Winkler</surname><given-names>Olga</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Burback</surname><given-names>Lisa</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Yanbo</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zaiane</surname><given-names>Osmar R</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Monson</surname><given-names>Candice</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff9">9</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sharma</surname><given-names>Divya</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff10">10</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Krishnan</surname><given-names>Sri</given-names></name><degrees>PEng, PhD</degrees><xref ref-type="aff" rid="aff11">11</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Greenshaw</surname><given-names>Andrew J</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zeifman</surname><given-names>Richard</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff12">12</xref><xref ref-type="aff" rid="aff13">13</xref><xref ref-type="aff" rid="aff14">14</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Selby</surname><given-names>Peter</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff15">15</xref><xref ref-type="aff" rid="aff16">16</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Bhat</surname><given-names>Venkat</given-names></name><degrees>MSc, MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff16">16</xref></contrib></contrib-group><aff id="aff1"><institution>AI for Mental Health Program, St. Michael&#x2019;s Hospital, Unity Health Toronto</institution><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><aff id="aff2"><institution>Temerty Centre for Artificial Intelligence Research and Education in Medicine, University of Toronto</institution><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><aff id="aff3"><institution>Campbell Family Mental Health Research Institute, Centre for Addiction and Mental Health</institution><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><aff id="aff4"><institution>Adult Neurodevelopment and Geriatric Psychiatry Division, Centre for Addiction and Mental Health</institution><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><aff id="aff5"><institution>Department of Psychiatry, University of Alberta</institution><addr-line>Edmonton</addr-line><addr-line>AB</addr-line><country>Canada</country></aff><aff id="aff6"><institution>Neuroscience and Mental Health Institute, University of Alberta</institution><addr-line>Edmonton</addr-line><addr-line>AB</addr-line><country>Canada</country></aff><aff id="aff7"><institution>Department of Computing Science, University of Alberta</institution><addr-line>Edmonton</addr-line><addr-line>AB</addr-line><country>Canada</country></aff><aff id="aff8"><institution>Alberta Machine Intelligence Institute</institution><addr-line>Edmonton</addr-line><addr-line>AB</addr-line><country>Canada</country></aff><aff id="aff9"><institution>Department of Psychology, Toronto Metropolitan University</institution><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><aff id="aff10"><institution>Department of Mathematics and Statistics, York University</institution><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><aff id="aff11"><institution>Department of Electrical, Computer, and Biomedical Engineering, Toronto Metropolitan University</institution><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><aff id="aff12"><institution>NYU Langone Center for Psychedelic Medicine, Department of Psychiatry, NYU Grossman School of Medicine</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff13"><institution>Center for Psychedelic Research, Department of Brain Sciences, Imperial College London</institution><addr-line>London</addr-line><country>United Kingdom</country></aff><aff id="aff14"><institution>Department of Psychology, The New School for Social Research</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff15"><institution>Department of Family and Community Medicine and Dalla Lana School of Public Health, University of Toronto</institution><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><aff id="aff16"><institution>Department of Psychiatry, University of Toronto</institution><addr-line>250 College Street, 8th Floor</addr-line><addr-line>Toronto</addr-line><addr-line>ON</addr-line><country>Canada</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Brini</surname><given-names>Stefano</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Chartash</surname><given-names>David</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Gaus</surname><given-names>Richard</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Venkat Bhat, MSc, MD, Department of Psychiatry, University of Toronto, 250 College Street, 8th FloorToronto, ON, M5T 1R8, Canada, 1 416-360-4000 ext 76404; <email>venkat.bhat@utoronto.ca</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>18</day><month>8</month><year>2026</year></pub-date><volume>12</volume><elocation-id>e92964</elocation-id><history><date date-type="received"><day>06</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>09</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>20</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Mohammad Amin Kamaleddin, Mina Mirjalili, Reza Barzegar, Nghia Trung Le, Zachary Cote, Olga Winkler, Lisa Burback, Yanbo Zhang, Osmar R Zaiane, Candice Monson, Divya Sharma, Sri Krishnan, Andrew J Greenshaw, Richard Zeifman, Peter Selby, Venkat Bhat. Originally published in JMIR Medical Education (<ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org">https://mededu.jmir.org</ext-link>), 18.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org/">https://mededu.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mededu.jmir.org/2026/1/e92964"/><abstract><sec><title>Background</title><p>Psychotherapy training is difficult to scale because manual rating of motivational interviewing (MI) and cognitive behavioral therapy (CBT) sessions is time-intensive, requires trained raters, and is subject to rater variability. Large language models (LLMs) may support simulation-based training and rubric-guided scoring, but early-stage evidence is needed before such systems can be applied to real learners.</p></sec><sec><title>Objective</title><p>This study aimed to conduct a simulation-based evaluation of a multiagent LLM framework for generating and scoring stylized MI and CBT training encounters.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a cross-sectional evaluation of a multiagent framework comprising Student, Patient, Evaluator, and Feedback agents. The Student agent conducted synthetic MI or CBT encounters with Patient agents derived from structured profiles. The Evaluator agent scored transcripts using study-specific MI and CBT scoring forms. Internal discrimination was tested across prompt-engineered novice, intermediate, and expert Student agent profiles using 133 MI and 102 CBT Patient profiles. Preliminary external grounding was assessed using 133 annotated motivational interviewing (AnnoMI) transcripts with coarse, metadata-derived, high/low session-level quality labels. Agreement with a pragmatic human-rater benchmark was evaluated using 16 independent human raters per modality. Criterion-level comparisons used paired Wilcoxon signed-rank tests with Benjamini-Hochberg false discovery rate correction at <italic>q</italic>&#x003C;0.05. Agreement analyses used intraclass correlation coefficients (ICCs) with bootstrap 95% CIs. A secondary prompt-augmentation sensitivity analysis tested whether appending criterion-referenced feedback text to the novice Student-agent prompt shifted subsequent Evaluator-assigned scores.</p></sec><sec sec-type="results"><title>Results</title><p>Evaluator scores increased across prompt-defined Student-agent competence levels. For MI, overall mean scores increased from 1.18 (95% CI 1.16&#x2010;1.20) for novice profiles to 1.75 (95% CI 1.69&#x2010;1.81) for intermediate profiles and to 3.57 (95% CI 3.48&#x2010;3.66) for expert profiles. For CBT, overall mean scores increased from 0.83 (95% CI 0.78&#x2010;0.88) to 2.18 (95% CI 2.10&#x2010;2.26) and 4.38 (95% CI 4.28&#x2010;4.48), respectively. Criterion-level planned contrasts were significant after false discovery rate correction. On AnnoMI transcripts, Evaluator scores aligned with metadata-derived high/low session labels, with 91.7% classification accuracy at the prespecified threshold. Human interrater reliability was an ICC(2,1) of 0.866 for MI and 0.769 for CBT. Evaluator-vs-human-consensus agreement was an ICC(2,1) of 0.959 for MI and 0.934 for CBT. In the secondary prompt-augmentation analysis, overall MI scores shifted from 1.18 to 1.44, and CBT scores shifted from 0.83 to 1.01.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This proof-of-concept study suggests that a rubric-guided multiagent LLM framework can score stylized synthetic MI and CBT transcripts along expected prompt-defined competence gradients and align with preliminary external and human-rater benchmarks. The study is innovative in separating synthetic learner, patient, scoring, and feedback-generation roles within a single simulation workflow, extending prior LLM work from plausible dialogue generation toward rubric-linked scoring. The findings support further development of scalable simulation tools for psychotherapy training research. Prospective studies with human trainees, standardized or real patients, and formal psychometric testing are required.</p></sec></abstract><kwd-group><kwd>psychotherapy</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>large language models</kwd><kwd>motivational interviewing</kwd><kwd>cognitive behavioral therapy</kwd><kwd>treatment fidelity</kwd><kwd>clinical competence</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Providing opportunities to practice clinical counseling or psychotherapy skills before seeing patients is foundational to psychiatric and psychological education [<xref ref-type="bibr" rid="ref1">1</xref>]. Traditional simulations using peers or trained standardized patients offer realistic yet controlled environments for skill acquisition and are particularly effective for practicing procedural and interpersonal skills [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. However, simulation with standardized patients is resource-intensive to create, staff, and scale across cohorts, sites, and time zones, demanding substantial educator time and expertise. These challenges include variability in rater expertise and feedback quality, as well as the high cost of sustained supervision [<xref ref-type="bibr" rid="ref5">5</xref>]. These constraints are especially salient for psychotherapy-focused encounters, where trainees must combine theoretical understanding of a therapeutic approach with repeated, coached practice to develop skillful listening, empathy, and collaborative problem-solving [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Recent reviews of virtual simulation in health professions education indicate that digital patient tools can support repeated, standardized communication-skills practice, but that evidence remains heterogeneous and many systems are still evaluated mainly on usability, satisfaction, or perceived realism rather than objective competency outcomes [<xref ref-type="bibr" rid="ref7">7</xref>]. The rapid emergence of large language model (LLM)-based virtual patients has intensified both this opportunity and this concern: recent scoping-review evidence suggests that LLM-based virtual-patient research is expanding quickly but remains early-stage, with inconsistent assessment designs and limited use of validated tools or objective learning outcomes [<xref ref-type="bibr" rid="ref8">8</xref>]. For psychotherapy education, this gap is especially important because competence is not captured by plausible dialogue alone; trainees must demonstrate model-specific behaviors such as reflective listening, evocation, autonomy support, collaborative agenda setting, and guided discovery [<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Motivational interviewing (MI) [<xref ref-type="bibr" rid="ref10">10</xref>] and cognitive behavioral therapy (CBT) [<xref ref-type="bibr" rid="ref11">11</xref>] are evidence-based interventions taught across mental health training programs. Both have structured, model-anchored therapist behaviors, making them relevant candidates for AI-based tools that support rubric-guided fidelity assessment. Observer-rated fidelity and competence instruments, such as the Motivational Interviewing Treatment Integrity (MITI) [<xref ref-type="bibr" rid="ref12">12</xref>] and Motivational Interviewing Skill Code (MISC) [<xref ref-type="bibr" rid="ref13">13</xref>] for MI and the Cognitive Therapy Rating Scale (CTRS) [<xref ref-type="bibr" rid="ref14">14</xref>] for CBT, provide structured scoring anchors for rating therapist behaviors and session structure from transcripts or recordings. However, applying these scoring tools consistently is labor-intensive and susceptible to rater drift.</p><p>Automating psychotherapy fidelity assessment has therefore become an important target for computational behavioral science [<xref ref-type="bibr" rid="ref15">15</xref>]. Recent MI-focused work has shown that AI models can classify counselor and client MI behaviors in chat-based counseling data and may support scalable coding, adherence monitoring, and personalized feedback [<xref ref-type="bibr" rid="ref16">16</xref>]. Other recent work has used LLM-based and sequential modeling approaches to estimate MI session quality, reinforcing the potential of automated transcript-level assessment while also underscoring the need for further validation against expert and field-collected data [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>LLMs add a complementary capability to prior virtual-patient and automated-coding approaches: they can generate interactive patient dialogue, instantiate different learner or patient roles, and produce structured explanations or feedback within a single workflow [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref21">21</xref>]. Agentic designs use multiple specialized agents to simulate roles, critique outputs, and enforce constraints, enabling scalable practice through role separation, cross-agent checking, and standardized evaluation workflows with feedback aligned to therapeutic model criteria [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. Recent LLM-based simulated-patient systems and agentic frameworks have shown feasibility for medical education, including scalable patient simulation, communication-skills practice, and automated feedback [<xref ref-type="bibr" rid="ref25">25</xref>]. However, many such systems remain oriented toward history taking, clinical reasoning, or generic communication practice rather than psychotherapy-specific fidelity assessment [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. In behavioral health, recent scholarship has emphasized that LLMs could support psychotherapy training and research, but that this is a high-stakes domain requiring transparent rubrics, careful evaluation, human oversight, and attention to safety, bias, and the distinction between superficially empathic language and model-concordant clinical practice [<xref ref-type="bibr" rid="ref28">28</xref>]. Emerging psychotherapy-specific studies have begun to evaluate AI-generated MI patient simulations and LLM-supported counseling feedback [<xref ref-type="bibr" rid="ref29">29</xref>], but evidence remains limited regarding rubric-linked scoring sensitivity, agreement with human-rater benchmarks, feedback alignment, and evaluation across more than one psychotherapy modality.</p><p>To address these gaps, we conducted a preclinical, simulation-based evaluation of a rubric-guided, multiagent LLM framework comprising (1) a Patient agent that enacts structured Patient profiles through case-consistent responses; (2) a Student agent that conducts the synthetic session; (3) an Evaluator agent that scores Student-agent behavior against explicit criteria; and (4) a Feedback agent that generates criterion-referenced narrative recommendations. MI scoring used a study-specific MITI or MISC-derived global-rating form, and CBT scoring used a CTRS-derived form. MI Patient profiles were generated from a manually constructed template across common behavior-change contexts, whereas CBT Patient profiles were reformatted from the <italic>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</italic> (<italic>DSM-5</italic>) Clinical Cases into a standardized Patient-agent prompt structure [<xref ref-type="bibr" rid="ref30">30</xref>]. Annotated motivational interviewing (AnnoMI) [<xref ref-type="bibr" rid="ref31">31</xref>] was used separately for a preliminary external comparison of Evaluator scores with coarse metadata-derived MI session-level quality labels. These simulations were intended as stylized benchmark examples for early-stage system evaluation, not as evidence that LLM-generated sessions reproduce the ecological complexity of real psychotherapy encounters.</p><p>The aim of this study was to conduct a preclinical, cross-sectional, simulation-based evaluation of a rubric-guided, multiagent LLM framework for psychotherapy training and fidelity assessment in MI and CBT. The primary objective was to determine whether the Evaluator agent could discriminate among prompt-defined novice, intermediate, and expert Student-agent performance across standardized synthetic Patient-agent encounters. We hypothesized that Evaluator-assigned scores would increase monotonically across these competence levels. Secondary objectives were to assess whether Evaluator-agent scores aligned with external AnnoMI session-level MI quality labels, to estimate agreement between Evaluator-agent scores and a pragmatic human-rater benchmark, and to examine whether appending criterion-referenced feedback to novice Student-agent prompts would shift subsequent Evaluator-assigned scores. Exploratory analyses examined Student-agent response-consistency metrics across prompt-defined competence levels.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This was a preclinical, cross-sectional, simulation-based evaluation of a multiagent LLM framework for psychotherapy training and fidelity scoring. The study used synthetic Patient-agent encounters and prompt-defined Student-agent profiles to evaluate (1) internal scoring sensitivity across prompt-defined competence levels, (2) preliminary external alignment with AnnoMI session-level labels, (3) agreement with a human-rater benchmark, (4) secondary prompt-augmentation sensitivity after feedback text was appended to the novice Student-agent prompt, and (5) exploratory Student-agent response-consistency metrics. Where applicable, reporting of the AI system, evaluation setting, and human comparators followed the DECIDE-AI (Developmental and Exploratory Clinical Investigations of Decision Support Systems Driven by AI) guideline for early-stage evaluation of AI decision-support systems (<xref ref-type="supplementary-material" rid="app2">Checklist 1</xref>) [<xref ref-type="bibr" rid="ref32">32</xref>].</p></sec><sec id="s2-2"><title>Setting</title><p>Simulations and analyses were conducted computationally using Python or a LangGraph (LangChain) workflow. Human raters completed transcript-rating tasks remotely using synthetic or deidentified transcripts.</p></sec><sec id="s2-3"><title>Inclusion and Exclusion</title><p>AnnoMI transcripts were included if they had an available high or low session-level quality label. Human-rater data were included if the rater completed all assigned transcript-rating fields.</p></sec><sec id="s2-4"><title>Sampling Procedures</title><p>The MI and CBT Patient-agent profile sets were assembled purposively to provide broad coverage of behavior-change topics and diagnostic presentations for synthetic simulation, rather than to represent an epidemiologic or clinical population sample. MI Patient-agent profiles were generated using a manually written template across common MI-relevant behavior-change contexts. CBT Patient-agent profiles were reformatted from <italic>DSM-5</italic> Clinical Cases into standardized Patient-agent prompts. AnnoMI transcripts were included as a publicly available external comparison dataset. Physician raters were recruited purposively based on their clinical experience with MI and CBT. Additional independent human raters were recruited through Prolific using convenience sampling from the participant pool. Eligibility criteria required English as a first language.</p></sec><sec id="s2-5"><title>Sample Size, Power, and Precision</title><p>The study size was determined pragmatically based on the available Patient-agent profile set, the eligible AnnoMI transcripts with session-level labels, and the feasibility of obtaining human ratings. We analyzed 133 MI Patient-agent profiles, 102 CBT Patient-agent profiles, 133 AnnoMI transcripts, and 10 selected sessions per modality for human rating by 16 raters. Precision was summarized with 95% CIs for key agreement metrics, classification metrics, and mean estimates.</p></sec><sec id="s2-6"><title>Participant Characteristics</title><p>No real patients or real learners participated in the simulation experiments. Human involvement was limited to independent transcript rating. For each modality, 16 human raters scored the selected transcripts. The rater panel included 3 practicing physicians with clinical experience using MI and CBT, with a median of 5 (IQR 3-5) years of relevant clinical practice. Thirteen additional independent human raters were recruited through Prolific. All raters provided informed consent and were compensated at a rate equivalent to US $20 per hour.</p></sec><sec id="s2-7"><title>Patient-Agent Profiles</title><p>For MI, we constructed 133 Patient-agent profiles from a manually written template informed by MITI [<xref ref-type="bibr" rid="ref12">12</xref>] and MISC [<xref ref-type="bibr" rid="ref13">13</xref>] concepts and common MI-relevant behavior-change scenarios, including alcohol use, smoking cessation, weight management, and medication adherence. One profile was manually written as a template specifying required sections: role-prompt field, demographic and presenting details, behavioral presentation, MI-relevant elements, mental-status cues, and simulation-prompt field. We then used single-example template prompting, meaning that the LLM was given one completed exemplar profile, the required output fields, and instructions to generate additional Patient-agent profiles in the same structure. MI profile generation used GPT-4.1-mini. Demographic and clinical variation was introduced through prompt-specified variation in age, sex or gender, cultural background, presenting problem, readiness to change, sustain talk, change talk, and ambivalence (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>For CBT, we constructed 102 Patient-agent profiles by reformatting cases from <italic>DSM-5</italic> Clinical Cases [<xref ref-type="bibr" rid="ref30">30</xref>] into a standardized Patient-agent format using GPT-4.1-mini. Each profile included a case synopsis, behavioral presentation, symptom list, structured patient history, and diagnosis. The profile set was intended to provide broad diagnostic and cognitive or behavioral diversity for synthetic CBT simulations rather than a representative sample of CBT patients (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-8"><title>AnnoMI External Comparison Dataset</title><p>We used 133 publicly available AnnoMI counseling transcripts for external comparison. AnnoMI includes expert utterance-level annotations; however, the present analysis did not use utterance-level behavior codes. Instead, we compared Evaluator-assigned global MI scores with the corpus-level high or low session quality labels, which are coarse, metadata-derived labels based on the original video context. This analysis was therefore treated as preliminary external grounding rather than as validation against expert fidelity coders.</p></sec><sec id="s2-9"><title>Measures and Covariates</title><p>Primary outcomes were criterion-level Evaluator scores within MI and CBT. MI scores were assigned across 9 criteria on a 1 to 5 scale using a study-specific MITI or MISC-derived global-rating form. CBT scores were assigned across 11 criteria on a 0 to 6 scale using a CTRS-derived scoring form [<xref ref-type="bibr" rid="ref14">14</xref>]. No demographic or clinical covariates were modeled because Patient-agent profiles were synthetic and were not designed as a representative clinical sample.</p></sec><sec id="s2-10"><title>Evaluator Agent Rubric Construction</title><p>For MI, we created a study-specific MITI or MISC-derived rubric rather than applying the full MITI or MISC instruments. The purpose was to provide a compact global-rating form suitable for short simulated transcripts. Items were selected if they represented core MI fidelity constructs and could be judged at the transcript level as global performance domains. The 9 selected domains were Cultivating Change Talk, Softening Sustain Talk, Partnership, Empathy, Acceptance, Direction, Autonomy Support, Collaboration, and Evocation. MITI-derived domains were retained as global 1 to 5 ratings. MISC-derived domains were harmonized into the same 1 to 5 ordinal global-rating format using the source construct definitions as behavioral anchors. MISC behavior-count codes were not used as count variables.</p><p>The MI scoring form used a 1 to 5 global-rating scale for each criterion (1=&#x201C;absent, clearly inconsistent with the construct, or strongly nonadherent&#x201D;; 2=&#x201C;minimal or mostly ineffective evidence&#x201D;; 3=&#x201C;partial, inconsistent, or mixed evidence&#x201D;; 4=&#x201C;generally competent and mostly consistent evidence&#x201D;; and 5=&#x201C;strong, sustained, and consistently adherent evidence across the transcript&#x201D;). Item-specific anchor descriptions are provided in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>For CBT, we used the 11 CTRS-derived domains using the original 0 to 6 structure (0=&#x201C;poor or absent evidence of the competency&#x201D;; 2=&#x201C;limited or partially adequate performance&#x201D;; 4=&#x201C;good or competent performance&#x201D;; and 6=&#x201C;excellent, sustained, and well-integrated performance&#x201D;). Scores of 1, 3, and 5 were used for intermediate judgments between adjacent anchors (Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>For both approaches, the Evaluator generated criterion-level scores linked to behavioral anchors, and the Feedback agent generated criterion-referenced narrative recommendations based on the transcript, scoring criteria, and scored evaluation output (Tables S5 and S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-11"><title>Narrative Feedback Generation and Alignment</title><p>The Evaluator agent assigned criterion-level scores. The Feedback agent then generated narrative recommendations using the transcript, scoring criteria, and the scored evaluation output. Feedback outputs were structured by criterion and included the assigned score, a transcript-grounded explanation, a short summary, and actionable recommendations. We treated narrative-feedback alignment as structural consistency between the numeric score, the criterion-specific explanation, and the recommendations provided. Representative aligned MI and CBT feedback excerpts are provided in Table S13 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-12"><title>Multiagent System and Model Configuration</title><p>We implemented a role-separated multiagent framework comprising Student, Patient, Evaluator, and Feedback agents to simulate MI and CBT encounters, score transcripts, and generate criterion-referenced narrative feedback. The system was implemented in Python version 3.11 and orchestrated using LangGraph to support stateful workflows. The Student and Patient agents used GPT-4.1-mini. The <sc>S</sc>tudent agent used a temperature of 0.5 to balance adherence to the prompt-defined competence profile with conversational flexibility, whereas the Patient agent used a temperature of 1.0 to allow greater variability in synthetic patient responses. The Evaluator agent used Gemini 2.5 Flash Lite with a temperature of 0.2 to promote more stable rubric scoring and to assign the scoring role to a different model family from the Student and Patient generators. Narrative recommendations were generated by a separate Feedback agent using Claude Haiku 4.5 with a temperature of 0.2. Full model versions, providers, routing endpoints, temperatures, and token settings are reported in Table S14 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>At runtime, the system loaded the selected approach and Patient-agent profile, executed a fixed-length exchange loop, and then called the Evaluator agent on the full transcript to generate criterion-level scores. The Feedback agent then generated criterion-referenced narrative recommendations.</p></sec><sec id="s2-13"><title>Student-Agent Conditions</title><p>To evaluate internal discrimination, we engineered 3 Student-agent prompts for each approach: novice, intermediate, and expert. These profiles were prompt-defined competence levels intended to generate stylized benchmark transcripts rather than represent the full distribution of real trainee performance. The novice, intermediate, and expert profiles were used to create controlled behavioral contrasts. This allowed us to test whether the Evaluator was sensitive to increases in prompt-defined rubric adherence under standardized Patient-agent and session-length conditions.</p><p>Prompts were self-contained and encoded the communication style, language patterns, and modality-specific behaviors. Each simulated session was capped at 10 therapist-patient exchanges, in which one exchange was defined as 1 Student-agent turn followed by 1 Patient-agent response. This fixed length was selected a priori as a pragmatic standardization constraint to keep dialogue length constant across Patient-agent profiles and Student-agent conditions, and to approximate a brief, focused training encounter rather than a full psychotherapy session.</p><p>The novice profile represented low rubric adherence, characterized by directive communication, limited reflection, weak collaboration, poor agenda setting, premature advice, or technique use, and limited use of modality-specific strategies. The intermediate profile represented partial and inconsistent rubric adherence, with some appropriate MI or CBT behaviors but recurrent omissions or poorly timed interventions. This condition was added to avoid relying solely on maximally separated novice or expert endpoints. The expert profile represented a high-adherence benchmark, with consistent use of modality-concordant behaviors such as reflective listening, autonomy support, evocation, collaborative agenda setting, Socratic questioning, case formulation, and structured homework planning.</p><p>Student-agent profiles were prompt-engineered AI agents, not human learners. The novice, intermediate, and expert conditions were designed to create stylized benchmark transcripts with expected differences in MI- or CBT-consistent behaviors. These conditions were used to test the internal Evaluator&#x2019;s sensitivity to prompt-defined competence levels. Student prompts for MI and CBT, including novice, intermediate, expert, and novice+feedback prompt variants, are provided in Tables S7&#x2013;S12 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>To prevent application-level carryover between trials, each Patient-profile &#x00D7; Student-condition run was executed in an independent session state. For each run, the system initialized empty interview and evaluator message histories, loaded the selected Patient-agent profile and Student-agent prompt, and generated a new transcript without access to prior transcripts, scores, feedback, or dialogue histories from other runs. Static resources, including model weights, model configurations, rubric prompts, and profile files, were reused as experimental inputs; however, session-level conversational state was not reused.</p></sec><sec id="s2-14"><title>Human Rater Scoring and Reliability Analysis</title><p>A pragmatic human-rater benchmark was assembled to compare Evaluator-agent scores with independent human ratings. For each modality, 16 human raters scored the selected transcripts. The panel included 3 practicing physicians with clinical experience using MI and CBT, with a median of 5 years of relevant clinical practice. Thirteen additional independent raters were recruited through Prolific. All raters provided informed consent and were compensated at a rate equivalent to US $20 per hour. Each rater independently scored 10 MI sessions across 9 MITI or MISC-derived criteria and 10 CBT sessions across eleven CTRS-derived criteria. Raters used the study-specific MI scoring form and the CTRS-derived CBT scoring form without access to other raters&#x2019; scores or Evaluator-agent outputs. Source manuals and item descriptions were provided as conceptual guidance.</p><p>Agreement analyses used session &#x00D7; criterion cells as the unit of analysis: 90 observations for MI and 110 observations for CBT. The intraclass correlation coefficient (ICC) (2,1) was used for 2-way random-effects, absolute-agreement, and single-rating reliability. Percentile-bootstrap 95% CIs were computed by resampling session &#x00D7; criterion observations. Because an Evaluator-vs-mean-consensus ICC is not structurally equivalent to a human interrater ICC, we also computed Evaluator-vs-individual-rater ICCs.</p></sec><sec id="s2-15"><title>Prompt-Augmentation Sensitivity Analysis</title><p>To assess whether criterion-referenced feedback text altered subsequent Student-agent behavior, we conducted a secondary prompt-augmentation sensitivity analysis. For each Patient-agent profile, the novice Student agent first completed a 10-exchange session. The recommendation section from the feedback report was then extracted and appended to the same novice Student-agent prompt before we ran a second 10-exchange session as a new, independent session state with the same Patient-agent profile. The underlying novice prompt was otherwise unchanged. This analysis was designed to test system responsiveness to feedback text within the simulation pipeline.</p></sec><sec id="s2-16"><title>Exploratory Student-Agent Dispersion Analysis</title><p>As an exploratory proxy for generative consistency, we analyzed token-level log probabilities from Student-agent responses. For each Student-agent turn, token log probabilities were extracted during GPT-4.1-mini generation, clipped at &#x2212;10 to reduce the influence of sentinel or near-zero-probability tokens, and averaged across tokens. The session-level Student-agent average log probability was then computed as the mean across the 10 Student-agent turns. For interpretability, this value was transformed to the geometric mean token probability: exp(the session average log probability) &#x00D7; 100. We summarized variability across Patient-agent profiles within each Student-agent condition using SD and IQR, with bootstrap CIs.</p></sec><sec id="s2-17"><title>Missing Data</title><p>Missingness was assessed for generated transcripts, Evaluator-assigned criterion scores, AnnoMI labels, and human-rater scores. No missing data were observed in the final analytic datasets; therefore, no imputation was performed.</p></sec><sec id="s2-18"><title>Data Analysis</title><p>Criterion-level scores were analyzed within MI (9 criteria, 1&#x2010;5 scale) and CBT (11 criteria, 0&#x2010;6 scale). Internal discrimination analyses compared novice, intermediate, and expert Student-agent conditions within Patient-agent profiles. Paired comparisons across Student-agent conditions and pre- or post-prompt-augmentation comparisons were conducted using 2-sided Wilcoxon signed-rank tests with a significance <italic>&#x03B1;</italic> level of .05. For criterion-level analyses, Benjamini-Hochberg false discovery rate (FDR) correction was applied within each modality and planned contrast family. Adjusted <italic>q</italic> values are reported for criterion-level comparisons, with <italic>q</italic>&#x003C;0.05 considered statistically significant. Overall-average comparisons were treated as separate planned summary tests and are reported with their corresponding <italic>P</italic> values, absolute mean changes, and effect sizes. Effect sizes were reported as rank-biserial correlations.</p><p>For mean scores and mean changes, 95% CIs were estimated using nonparametric bootstrap resampling across Patient-agent profiles. For the AnnoMI external comparison, Evaluator-assigned global MI scores were compared with coarse metadata-derived high or low session-level quality labels. Scores less than or equal to the prespecified threshold were classified as low quality, and scores greater than the threshold were classified as high quality. Accuracy, precision, recall, and <italic>F</italic><sub>1</sub>-score were computed with high quality as the positive class and reported with 95% CIs. Human-rater reliability analyses used ICC(2,1), weighted Cohen &#x03BA;, mean within-cell SD, mean absolute error, root mean square error, Pearson correlation, and Spearman correlation, as applicable. All statistical analyses were performed using Python version 3.11. Nonparametric procedures were selected because rubric scores were ordinal and potentially nonnormally distributed.</p></sec><sec id="s2-19"><title>Ethical Considerations</title><p>This work was conducted as a technical feasibility and simulation-based benchmarking study of a multiagent LLM framework prior to future controlled educational or clinical evaluation studies. The simulation and secondary-data components used synthetic agent-generated transcripts and deidentified source materials. No real patients or real learners participated; no identifiable patient information was used; and no clinical decisions were informed by the study outputs. In accordance with the Tri-Council Policy Statement: Ethical Conduct for Research Involving Humans&#x2013;TCPS 2 [<xref ref-type="bibr" rid="ref33">33</xref>], the simulation and secondary-data components were determined not to require research ethics board review.</p><p>Human involvement was limited to the independent evaluation of synthetic or deidentified transcripts. Raters were informed about the purpose and procedures of the rating task, provided informed consent before participation, and were compensated at a rate equivalent to US $20 per hour. Raters did not interact with patients, learners, or clinical systems. No personal health information or sensitive clinical information was collected from raters. Rater data were stored and analyzed in deidentified form and reported only in aggregate. Formal Research Ethics Board review was not sought because this work was classified as minimal-risk technical feasibility and simulation-based benchmarking research. This classification was based on the use of synthetic or deidentified transcripts, the limited scope of human-rater involvement, the absence of patient or learner interaction, the absence of personal health information or sensitive participant data, deidentified analysis, and aggregate reporting only. Future studies designed to evaluate real learners, standardized or real patients, educational effectiveness, clinical performance, or identifiable participant information will undergo formal Research Ethics Board review prior to implementation.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview of the Evaluation Framework</title><p>The evaluation consisted of fixed-length synthetic encounters between a Student agent and structured Patient agents in 2 psychotherapeutic approaches, MI and CBT, with Student-agent performance scored by an Evaluator agent using predefined rubric criteria (<xref ref-type="fig" rid="figure1">Figure 1</xref>). For MI, the Evaluator used a study-specific, 9-criterion MITI or MISC&#x2013;derived global-rating form on a 1 to 5 scale (Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For CBT, the Evaluator applied an 11-item CTRS&#x2013;derived form on a 0 to 6 scale (Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). All sessions used the fixed 10-exchange format defined in the &#x201C;Methods&#x201D; section. Scores were averaged across 133 MI and 102 CBT Patient-agent profiles for each of the 3 prompt-defined Student-agent competence profiles: novice, intermediate, and expert.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Multiagent large language model framework for stylized psychotherapy-training simulation and rubric-based evaluation. The system comprises role-separated Student, Patient, Evaluator, and Feedback agents coordinated within a structured dialogue workflow. The Student agent, instantiated with GPT-4.1-mini, conducts the synthetic psychotherapy encounter based on a predefined prompt. The Patient agent, also instantiated with GPT-4.1-mini, enacts a structured Patient-agent profile and generates case-consistent responses during the fixed-length dialogue. The Evaluator agent, instantiated with Gemini 2.5 Flash Lite, scores the completed transcript against predefined evaluation criteria. The Feedback agent, instantiated with Claude Haiku 4.5, generates criterion-referenced narrative recommendations based on the transcript, scoring criteria, and the scored evaluation output. Feedback text could be appended to the novice Student-agent prompt in a secondary prompt-augmentation sensitivity analysis. Full model configurations, temperatures, and routing details are reported in Table S14 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e92964_fig01.png"/></fig></sec><sec id="s3-2"><title>Internal Discrimination Across Prompt-Defined Competence Levels</title><p>Evaluator scores showed graded increases across the 3 prompt-defined student-agent competence conditions. For MI, the overall mean score increased from 1.18 (95% CI 1.16&#x2010;1.20) for the novice profile to 1.75 (95% CI 1.69&#x2010;1.81) for the intermediate profile and to 3.57 (95% CI 3.48&#x2010;3.66) for the expert profile (<xref ref-type="fig" rid="figure2">Figure 2A</xref>). For CBT, the overall mean score increased from 0.83 (95% CI 0.78&#x2010;0.88) to 2.18 (95% CI 2.10&#x2010;2.26) and 4.38 (95% CI 4.28&#x2010;4.48) for the intermediate and expert profiles, respectively (<xref ref-type="fig" rid="figure2">Figure 2B</xref>). All criterion-level planned contrasts across student-agent competence conditions remained significant after Benjamini-Hochberg false discovery rate correction (<italic>q</italic>&#x003C;0.05). These findings support the Evaluator agent&#x2019;s sensitivity to expected differences across stylized prompt-defined competence levels, but should be interpreted as an internal scoring-sensitivity benchmark rather than evidence of real-world learner assessment.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Evaluator-scored performance across prompt-defined novice, intermediate, and expert Student-agent profiles. (A) Average criterion-level scores for motivational interviewing (MI) across 133 simulated Patient-agent profiles, evaluated using a study-specific Motivational Interviewing Treatment Integrity or Motivational Interviewing Skill Code (MITI/MISC)&#x2013;derived 1 to 5 global-rating form. (B) Average criterion-level scores for cognitive behavioral therapy (CBT) across 102 simulated Patient-agent profiles, evaluated using a Cognitive Therapy Rating Scale (CTRS)&#x2013;derived 0 to 6 scoring form. Bars show mean scores and error bars show the SE of the mean. Scores demonstrate graded separation across prompt-engineered competence levels. Criterion-level significance testing used paired Wilcoxon signed-rank tests with Benjamini-Hochberg false discovery rate correction within each modality and planned contrast; adjusted <italic>q</italic> values were used to determine significance. These analyses evaluate internal scoring sensitivity across stylized prompt-defined Student-agent profiles.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e92964_fig02.png"/></fig></sec><sec id="s3-3"><title>External Comparison Using AnnoMI Session-Level Quality Labels</title><p>We next examined whether Evaluator scores aligned with external session-level quality labels in the AnnoMI corpus (<xref ref-type="fig" rid="figure3">Figure 3</xref>). AnnoMI contains expert utterance-level annotations [<xref ref-type="bibr" rid="ref34">34</xref>], but this analysis used only the coarse high or low session-level quality labels associated with the source videos. Metadata-labeled high-quality sessions clustered primarily at Evaluator scores of 3 to 5, whereas metadata-labeled low-quality sessions clustered primarily at scores of 1 to 2. Using a threshold of 2, where scores &#x2264;2 were classified as low quality and scores &#x003E;2 were classified as high quality, yielded 91.7% accuracy, 95.4% precision, 94.5% recall, and an <italic>F</italic><sub>1</sub>-score of 95.0%.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>External comparison of Evaluator-agent motivational interviewing scores with annotated motivational interviewing (AnnoMI) metadata-derived quality labels. (A) Distribution of Evaluator-assigned motivational interviewing scores stratified by metadata-derived high or low session-level quality labels in the AnnoMI corpus. Bars show the percentage of sessions within each quality-label group receiving each score: low label (n=23) and high label (n=110). (B) Classification performance across Evaluator-score thresholds. Scores less than or equal to the threshold were classified as low quality, and scores greater than the threshold were classified as high quality. Performance metrics were computed with high quality as the positive class. A threshold of 2 yielded accuracy of 91.7%, precision of 95.4%, recall of 94.5%, and <italic>F</italic><sub>1</sub>-score of 95.0%. These metrics reflect alignment with coarse metadata-derived session-level labels.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e92964_fig03.png"/></fig></sec><sec id="s3-4"><title>Agreement Between Evaluator Scores and Human Raters</title><p>We compared Evaluator-agent scores with ratings from a pragmatic 16-rater human benchmark (<xref ref-type="fig" rid="figure4">Figures 4</xref> and <xref ref-type="fig" rid="figure5">5</xref>). Raters independently scored 10 selected MI sessions across 9 MITI- or MISC-derived criteria and 10 selected CBT sessions across 11 CTRS-derived criteria, yielding 90 and 110 session &#x00D7; criterion observations, respectively. Session-level heatmaps showed broadly similar, though not identical, rating patterns across the human-rater consensus and Evaluator-agent scores in both approaches.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Human rater reliability and Evaluator-agent agreement with individual raters and the human-rater consensus for motivational interviewing (MI) scoring. Sixteen human raters per modality independently scored selected transcripts using the study-specific scoring forms. (A) Evaluator-agent agreement with each individual human rater for motivational interviewing (MI), shown as the intraclass correlation coefficient (ICC; 2,1) with 95%-bootstrap CIs between 90 session &#x00D7; criterion observations. (B) Human interrater reliability for MI across 16 human raters. (C) MI heatmaps comparing the mean human-rater consensus with Evaluator-agent scores. (D) Evaluator-agent agreement with the MI human-rater consensus. MAE: mean absolute error; RMSE: root mean square error.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e92964_fig04.png"/></fig><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Human rater reliability and Evaluator-agent agreement with individual raters and the human-rater consensus for cognitive behavioral therapy (CBT) scoring. Sixteen human raters per modality independently scored selected transcripts using the study-specific scoring forms. (A) Evaluator-agent agreement with each individual human rater for cognitive behavioral therapy (CBT), shown as ICC(2,1) with 95% percentile-bootstrap CIs across 110 session &#x00D7; criterion observations. (B) Human interrater reliability for CBT across 16 human raters. (C) CBT heatmaps comparing the mean human-rater consensus with Evaluator-agent scores. (D) Evaluator-agent agreement with the CBT human-rater consensus. ICC: intraclass correlation coefficient; MAE: mean absolute error; RMSE: root mean square error.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e92964_fig05.png"/></fig><p>Human interrater reliability was good for MI, with an ICC(2,1) of 0.866 (95% CI 0.825&#x2010;0.898) across 90 session &#x00D7; criterion observations, a mean weighted &#x03BA; of 0.737, a mean SD per observation of 0.552, and a median SD per observation of 0.500. For CBT, human interrater reliability was an ICC(2,1) of 0.769 (95% CI 0.732&#x2010;0.799) across 110 observations, a mean weighted &#x03BA; of 0.550, a mean SD per observation of 1.014, and a median SD per observation of 0.999.</p><p>Evaluator-vs-human-consensus agreement was high for MI, with ICC(2,1) of 0.959 (95% CI 0.940&#x2010;0.973), Pearson <italic>r</italic> of 0.969, Spearman &#x03C1; of 0.878, mean absolute error (MAE) of 0.366, and root mean square error (RMSE) of 0.490. For CBT, Evaluator-vs-human-consensus agreement was ICC(2,1) of 0.934 (95% CI 0.913&#x2010;0.950), Pearson <italic>r</italic> of 0.948, Spearman &#x03C1; of 0.877, MAE of 0.622, and RMSE of 0.765.</p><p>Because agreement with an averaged human consensus is not directly comparable with interrater reliability among individual humans, we additionally computed Evaluator-vs-individual-rater ICCs. These ranged from 0.761 to 0.964 for MI and from 0.669 to 0.950 for CBT. In leave-one-human-out analyses, individual human raters compared with the mean of the remaining human raters showed median ICCs of 0.944 for MI and 0.877 for CBT. These analyses contextualize the Evaluator agent&#x2019;s agreement with both individual raters and consensus ratings.</p></sec><sec id="s3-5"><title>Impact of Evaluator Feedback on Novice Performance</title><p>In the prompt-augmentation sensitivity analysis, appending feedback recommendations to the novice Student-agent prompt produced modest score shifts in subsequent synthetic sessions (<xref ref-type="fig" rid="figure6">Figure 6</xref>). For MI, the overall average increased from 1.18 to 1.44, corresponding to an absolute mean change of +0.26 points (95% CI 0.19&#x2010;0.33; <italic>P</italic>&#x003C;.001; rank-biserial effect size=0.73). All 9 MI criteria showed significant increases after false discovery rate correction (<xref ref-type="fig" rid="figure6">Figure 6A</xref>). For CBT, the overall average increased from 0.83 to 1.01, corresponding to an absolute mean change of +0.18 points (95% CI 0.11&#x2010;0.25; <italic>P</italic>&#x003C;.001; rank-biserial effect size=0.55). A total of 9 of 11 CBT criteria showed significant increases after false discovery rate correction; Interpersonal Effectiveness and Focusing on Key Cognitions or Behaviors were not significant (<xref ref-type="fig" rid="figure6">Figure 6B</xref>). These findings indicate that feedback-text prompt augmentation shifted evaluator-assigned scores within the synthetic pipeline.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Prompt-augmentation sensitivity analysis using feedback text appended to the novice Student-agent prompt. (A) Motivational interviewing (MI) criterion-level scores before and after appending feedback recommendations to the novice Student-agent prompt across 133 Patient-agent profiles. The overall average increased from 1.18 to 1.44, corresponding to an absolute change of +0.26 points (<italic>P</italic>&#x003C;.001) and a rank-biserial effect size of 0.73. All 9 MI criteria showed significant increases after false discovery rate correction. (B) Cognitive behavioral therapy (CBT) criterion-level scores before and after feedback-text prompt augmentation across 102 Patient-agent profiles. The overall average increased from 0.83 to 1.01, corresponding to an absolute change of +0.18 points (<italic>P</italic>&#x003C;.001) and a rank-biserial effect size of 0.55. A total of 9 of 11 CBT criteria showed significant increases after false discovery rate correction. Asterisks denote Benjamini-Hochberg false discovery rate&#x2013;adjusted criterion-level significance: *<italic>q</italic>&#x003C;0.05, **<italic>q</italic>&#x003C;0.01, ***<italic>q</italic>&#x003C;0.001; n.s.=not significant after false discovery rate correction. Overall-average <italic>P</italic> values in the tables correspond to separate planned paired Wilcoxon signed-rank tests.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e92964_fig06.png"/></fig><p>Narrative feedback outputs were criterion-referenced and score-linked. In representative MI outputs, low scores in Acceptance, Empathy, Partnership, Autonomy Support, and Evocation were accompanied by explanations citing judgmental language, lack of reflective listening, directive advice, limited collaboration, and failure to elicit change talk. Corresponding recommendations emphasized avoiding confrontation, using complex reflections, eliciting patient values, and supporting patient choice. In representative CBT outputs, weak Agenda, Conceptualization, Guided Discovery, and Homework scores were paired with recommendations for collaborative agenda setting, psychoeducation, Socratic questioning, and structured between-session practice. These examples illustrate structural alignment between criterion scores, explanations, and recommendations.</p></sec><sec id="s3-6"><title>Exploratory Response-Consistency Analysis</title><p><xref ref-type="fig" rid="figure7">Figure 7</xref> shows the dispersion of Student-agent response log-probability-derived average token probability across Patient-agent profiles. Dispersion decreased across prompt-defined competence levels. For MI, the SD decreased from 4.99 percentage points (95% CI 4.19&#x2010;5.77) for the novice profile to 2.46 (95% CI 2.19&#x2010;2.70) for the intermediate profile to 1.74 (95% CI 1.53&#x2010;1.95) for the expert profile; the corresponding IQR decreased from 6.01 (95% CI 4.50&#x2010;7.76) to 3.57 (95% CI 2.96&#x2010;4.22) to 2.34 (95% CI 1.75&#x2010;3.00). For CBT, the SD decreased from 3.13 (95% CI 2.51&#x2010;3.70) to 2.46 (95% CI 2.08&#x2010;2.82) to 1.44 (95% CI 1.28&#x2010;1.60), and the IQR decreased from 4.06 (95% CI 2.98&#x2010;4.80) to 3.25 (95% CI 2.55&#x2010;4.10) to 2.46 (95% CI 1.97&#x2010;2.69). These findings suggest that higher-adherence Student-agent prompts produced more constrained and consistent responses across Patient-agent profiles. This exploratory analysis is an indirect proxy for generative consistency and does not directly measure hallucination, clinical realism, or safety.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Exploratory Student-agent log-probability dispersion across prompt-defined competence levels. For each simulated session, Student-agent token log probabilities were averaged across generated tokens and turns, then transformed into geometric mean token probability percentages. (A) SD of the session-level average probability across Patient-agent profiles for novice, intermediate, and expert Student-agent conditions in motivational interviewing (MI) and cognitive behavioral therapy (CBT). (B) IQR of the same measure. Error bars indicate bootstrap CIs. Lower dispersion indicates more consistent model response probabilities across Patient-agent profiles. Expert Student-agent responses showed lower dispersion than novice responses in both MI and CBT.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e92964_fig07.png"/></fig><p>A prototype user interface was developed to demonstrate how the multiagent workflow could be presented to end users, including approach selection, text- or voice-based practice, and display of criterion-level scores with narrative feedback (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This interface was not used in the present evaluation analyses and should be interpreted only as an implementation demonstration. Overall, the results provide early-stage evidence that a rubric-guided Evaluator agent can score stylized synthetic MI and CBT transcripts along expected prompt-defined competence gradients, align with a preliminary external comparison and a pragmatic human-rater benchmark, and show modest score shifts after feedback-text prompt augmentation.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings: Scalable and Interpretable Assessment of Psychotherapeutic Competence</title><p>Integration of LLMs into health professions education may transform how trainees acquire and practice complex communication skills at scale [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. In this preclinical, cross-sectional, simulation-based evaluation, we assessed a rubric-guided, multiagent LLM framework for stylized MI and CBT training simulations. The Evaluator agent scored synthetic transcripts in the expected direction across prompt-defined novice, intermediate, and expert Student-agent profiles; showed preliminary alignment with coarse AnnoMI session-level quality labels; aligned with a pragmatic human-rater benchmark; and showed modest score shifts after feedback-text prompt augmentation. The exploratory log-probability dispersion analysis further suggested that higher-adherence Student-agent prompts produced more constrained and consistent responses across Patient-agent profiles. By coupling standardized Patient simulations with an Evaluator that maps behaviors to established instruments, the system provides formative and summative signals that are interpretable within recognized competency frameworks [<xref ref-type="bibr" rid="ref35">35</xref>]. Because the agents operate over explicit rubrics and supply narrative justifications, the approach is amenable to audit and standard setting and offers a path to consistent assessment across sites and cohorts [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>].</p></sec><sec id="s4-2"><title>Interpretation and Comparison With Prior Work</title><p>The main innovation of this work is the role-separated architecture, in which Patient, Student, Evaluator, and Feedback agents serve distinct functions within one simulation workflow [<xref ref-type="bibr" rid="ref25">25</xref>]. Prior LLM studies in health professions education have largely emphasized plausible simulated conversations or tutoring interactions [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. This study extends that work by linking synthetic psychotherapy encounters to explicit scoring criteria, external comparison data, pragmatic human-rater agreement, and criterion-referenced feedback outputs [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. The contribution is therefore not that the system reproduces real psychotherapy, but that it provides an auditable framework for testing whether an LLM-based Evaluator responds to controlled, rubric-relevant differences in synthetic transcripts.</p><p>The internal discrimination analysis is important because a scoring system intended for psychotherapy training should, at minimum, assign higher scores when transcripts contain more rubric-concordant therapist behaviors [<xref ref-type="bibr" rid="ref38">38</xref>]. The graded pattern across novice, intermediate, and expert Student-agent profiles supports this internal scoring sensitivity. However, these Student-agent profiles were engineered competence archetypes and should not be interpreted as empirical representations of medical students, therapists in training, or expert clinicians. Future studies must test whether similar scoring patterns hold in human learners across defined training stages [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>The human-rater comparison provides a complementary benchmark for interpreting the Evaluator&#x2019;s scores. Although the rater panel was not composed of calibrated fidelity coders, Evaluator scores aligned closely with the average human-rater consensus and showed similar patterns when compared with individual raters. This suggests that, within the constrained setting of stylized synthetic transcripts and study-specific scoring forms, the Evaluator&#x2019;s rubric-based judgments approximated the ratings assigned by independent human raters. This finding is consistent with prior work suggesting that LLM-based evaluators can approximate human ratings [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. However, this agreement should not be interpreted as evidence of superiority over human raters or as a substitute for validation against trained fidelity coders.</p><p>Some criteria, particularly Empathy and Autonomy Support, require careful interpretation in LLM-agent simulations. In this study, Empathy was scored as observable language indicating understanding, validation, and reflective response to the patient&#x2019;s stated experience. Autonomy Support was scored as noncoercive, choice-supporting language consistent with MI principles. These are transcript-level behavioral ratings, not evidence that the LLM possesses affective empathy, clinical attunement, or the capacity to form a therapeutic relationship. Prior work on LLM empathy has similarly emphasized the distinction between responses perceived as empathic and genuine human empathy [<xref ref-type="bibr" rid="ref41">41</xref>]. We therefore interpret these scores as evidence that the Evaluator detected rubric-concordant language patterns in synthetic transcripts, not as evidence of authentic empathic capacity.</p><p>The narrative feedback component also requires cautious interpretation. Feedback outputs were criterion-referenced and structurally aligned with scores, explanations, and recommendations. This structure is useful for formative simulation design because it makes the basis for feedback explicit and reviewable [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. However, narrative feedback alignment was assessed qualitatively and illustratively in this study; it was not independently validated by blinded experts for accuracy, clinical appropriateness, safety, or educational utility [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. Similarly, the prompt-augmentation analysis showed that appending feedback text shifted subsequent Student-agent scores, but it does not demonstrate feedback efficacy, human learning, or skill transfer [<xref ref-type="bibr" rid="ref42">42</xref>].</p></sec><sec id="s4-3"><title>Limitations and Future Directions</title><p>A central limitation is that the simulated transcripts were stylized outputs of prompt-engineered agents rather than interactions involving real patients or human trainees. We did not compare generated dialogues with real psychotherapy sessions, and we did not test whether Patient or Student agents reproduced the behavior of actual patients or trainees. This design supported controlled benchmarking but limited assessment of flexible adaptation to complex, comorbid, culturally nuanced, or diagnostically ambiguous presentations [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. Similarly, the fixed 10-exchange session length standardized exposure across conditions but restricted observation of longer-form psychotherapy processes such as alliance development, rupture repair, homework review, and multisession change [<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref46">46</xref>]. Accordingly, the findings should be interpreted as evidence that the Evaluator can score controlled synthetic examples along expected rubric-defined gradients, not as evidence of ecological validity or readiness for summative assessment of real learners [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref47">47</xref>].</p><p>The external and human-rating benchmarks also have important limitations. Although AnnoMI contains expert utterance-level annotations [<xref ref-type="bibr" rid="ref31">31</xref>], this study used only coarse high or low session-level quality labels derived from video metadata and source context; these labels are not equivalent to expert transcript-level MI fidelity ratings. In addition, because AnnoMI is publicly available and predates the Evaluator model&#x2019;s knowledge cutoff, training-data contamination cannot be excluded [<xref ref-type="bibr" rid="ref48">48</xref>]. The human-rating analysis should likewise be interpreted as a pragmatic benchmark rather than a fidelity-coding reference standard. Although the rater panel provided useful comparison data, raters were not formally calibrated MITI, MISC, or CTRS fidelity coders. Future validation should therefore include trained and calibrated fidelity coders, predefined reliability thresholds, full fidelity-coding procedures, and prospectively collected transcripts from human trainees interacting with standardized or real patients.</p><p>The prompt-augmentation and feedback analyses should not be interpreted as evidence of educational efficacy. No human learner received feedback, practiced with the system, demonstrated skill retention, or transferred skills to a new clinical task. Instead, the postfeedback Student-agent prompt contained feedback recommendations appended to an otherwise novice prompt, meaning that observed score shifts may reflect LLM instruction-following or prompt sensitivity rather than learning [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>]. Narrative feedback outputs were criterion-referenced and structurally aligned with scores, explanations, and recommendations, but they were not independently validated by blinded experts for accuracy, clinical appropriateness, safety, or educational utility. Future studies should test feedback prospectively in human trainees, using independent feedback delivery, delayed reassessment, blinded expert ratings, and evaluation on unseen standardized-patient or real-patient encounters [<xref ref-type="bibr" rid="ref51">51</xref>-<xref ref-type="bibr" rid="ref53">53</xref>].</p><p>Finally, several implementation and safety limitations remain. LLMs are susceptible to hallucination, misinterpretation, unsafe recommendations, bias, and calibration drift [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref55">55</xref>]. This study did not formally adjudicate or quantify failure events such as hallucinated clinical details, internally inconsistent Patient-agent responses, or clinically inappropriate recommendations. The exploratory log-probability dispersion analysis provides only an indirect proxy for generative consistency and does not directly measure clinical realism, therapeutic safety, or fidelity to patient complexity. Because the simulations were text-only, the system could not evaluate tone, pacing, prosody, facial affect, gestures, silence, turn timing, or other nonverbal and paralinguistic behaviors that contribute to alliance and clinical communication [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref57">57</xref>]. Future versions should incorporate structured safety monitoring, expert adjudication of failure modes, multimodal inputs such as audio or video, and validation by domain-specific clinical experts [<xref ref-type="bibr" rid="ref56">56</xref>]. Model selection is also implementation-dependent: using different model families reduces direct same-model self-evaluation, but it does not eliminate shared pretraining exposure, model bias, or calibration drift [<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref59">59</xref>]. These issues should be addressed before the system is used beyond low-stakes formative simulation or research settings.</p></sec><sec id="s4-4"><title>Conclusions</title><p>This proof-of-concept study suggests that a rubric-guided multiagent LLM framework can score stylized synthetic MI and CBT transcripts along expected prompt-defined competence gradients and align with preliminary external and pragmatic human-rater benchmarks. The study is innovative in separating synthetic patient simulation, synthetic learner behavior, rubric-based scoring, and narrative feedback generation within a single auditable workflow. It differs from prior LLM simulation studies by moving beyond face-valid dialogue generation toward structured, criterion-linked evaluation. The findings support further development of scalable simulation tools for psychotherapy training research, fidelity-methods development, and low-stakes formative practice.</p></sec></sec></body><back><ack><p>The authors thank Martin Ivanov for his contributions to the development of this work. The authors declare the use of generative AI in the research and writing process. According to the Generative AI Delegation Taxonomy (GAIDeT, 2025), the following tasks were delegated to generative AI tools under full human supervision: code optimization, proofreading, and editing. The generative AI tool used was ChatGPT-5.5. Responsibility for the final manuscript lies entirely with the authors. Generative AI tools are not listed as authors and do not bear responsibility for the final outcomes. All substantive scientific content, analyses, interpretations, citations, manuscript revisions, and reviewer responses were reviewed, edited, and approved by the authors, who take full responsibility for the integrity and accuracy of the final work.</p></ack><notes><sec><title>Funding</title><p>This work was supported by scholarships or fellowships from Mitacs (Accelerate Fellowship), the Brain Canada Foundation (Rising Stars Award), and the Canadian Institutes of Health Research (CIHR; Canada Postdoctoral Research Award) to MAK.</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during the current study are available on GitHub [<xref ref-type="bibr" rid="ref60">60</xref>]. The repository provides synthetic motivational interviewing patient profiles, student prompts, analysis code, and structured cognitive behavioral therapy patient-agent attributes reformatted from the <italic>Diagnostic and Statistical Manual of Mental Disorders</italic>, <italic>Fifth Edition</italic> (<italic>DSM-5</italic>) Clinical Cases. Annotated motivational interviewing transcripts are not redistributed and can be accessed through the original dataset source.</p></sec></notes><fn-group><fn fn-type="con"><p>Concept and design: MAK, MM, VB</p><p>Acquisition, analysis, and interpretation of data: MAK, MM, RB, NTL, ZC</p><p>Drafting of the manuscript: MAK, MM</p><p>Critical review of the manuscript: MAK, MM, RB, NTL, ZC, OW, LB, YZ, ORZ, CM, DS, SK, AJG, RZ, PS, VB</p><p>Administrative, technical, and material support: MAK, VB</p><p>All authors contributed to and approved the final version of the manuscript</p></fn><fn fn-type="conflict"><p>MAK has received funding from Sanofi. LB, OW, and YZ were supported by the Academic Medicine and Health Services Program. RZ has received funding from the NYU Langone Psychedelic Medicine Research Training Program (funded by MindMed) and the Canadian Institutes of Health Research. PS has received funding, honoraria, or consulting fees from Pfizer, Shoppers Drug Mart, Bhasin Consulting Fund Inc, the Patient-Centered Outcomes Research Institute, AbbVie, Bristol-Myers Squibb, Evidera Inc, Johnson &#x0026; Johnson Group of Companies, Medcan Clinic, Inflexxion Inc, V-CC Systems Inc, MedPlan Communications, Kataka Medical Communications, Miller Medical Communications, Nvision Insight Group, and Sun Life Financial. VB has received research support from the Canadian Institutes of Health Research, the Brain and Behavior Foundation, the Ontario Ministry of Health Innovation Funds, the Royal College of Physicians and Surgeons of Canada, the Department of National Defence (Government of Canada), the New Frontiers in Research Fund, Associated Medical Services Inc Healthcare, the American Foundation for Suicide Prevention, Roche, Novartis, and Eisai. The funders had no role in the design, data collection, analysis, interpretation, or writing of this manuscript.</p></fn></fn-group><glossary><title>Abbreviations:</title><def-list><def-item><term id="abb1">AnnoMI</term><def><p>annotated motivational interviewing</p></def></def-item><def-item><term id="abb2">CBT</term><def><p>cognitive behavioral therapy</p></def></def-item><def-item><term id="abb3">CTRS</term><def><p>Cognitive Therapy Rating Scale</p></def></def-item><def-item><term id="abb4">DECIDE-AI</term><def><p>Developmental and Exploratory Clinical Investigations of Decision Support Systems Driven by AI</p></def></def-item><def-item><term id="abb5"><italic>DSM-5</italic></term><def><p><italic>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</italic></p></def></def-item><def-item><term id="abb6">FDR</term><def><p>false discovery rate</p></def></def-item><def-item><term id="abb7">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb8">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb9">MAE</term><def><p>mean absolute error</p></def></def-item><def-item><term id="abb10">MI</term><def><p>motivational interviewing</p></def></def-item><def-item><term id="abb11">MISC</term><def><p>Motivational Interviewing Skill Code</p></def></def-item><def-item><term id="abb12">MITI</term><def><p>Motivational Interviewing Treatment Integrity</p></def></def-item><def-item><term id="abb13">RMSE</term><def><p>root mean square error</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nurse</surname><given-names>K</given-names> </name><name name-style="western"><surname>O&#x2019;Shea</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ling</surname><given-names>M</given-names> </name><name name-style="western"><surname>Castle</surname><given-names>N</given-names> </name><name name-style="western"><surname>Sheen</surname><given-names>J</given-names> </name></person-group><article-title>The influence of deliberate practice on skill performance in therapeutic practice: a systematic review of early studies</article-title><source>Psychother Res</source><year>2025</year><month>03</month><volume>35</volume><issue>3</issue><fpage>353</fpage><lpage>367</lpage><pub-id pub-id-type="doi">10.1080/10503307.2024.2308159</pub-id><pub-id pub-id-type="medline">38295223</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elendu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Amaechi</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Okatta</surname><given-names>AU</given-names> </name><etal/></person-group><article-title>The impact of simulation-based training in medical education: a review</article-title><source>Medicine (Baltimore)</source><year>2024</year><month>07</month><day>5</day><volume>103</volume><issue>27</issue><fpage>e38813</fpage><pub-id pub-id-type="doi">10.1097/MD.0000000000038813</pub-id><pub-id pub-id-type="medline">38968472</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beal</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Kinnear</surname><given-names>J</given-names> </name><name name-style="western"><surname>Anderson</surname><given-names>CR</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Wamboldt</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hooper</surname><given-names>L</given-names> </name></person-group><article-title>The effectiveness of medical simulation in teaching medical students critical care medicine: a systematic review and meta-analysis</article-title><source>Simul Healthc</source><year>2017</year><month>04</month><volume>12</volume><issue>2</issue><fpage>104</fpage><lpage>116</lpage><pub-id pub-id-type="doi">10.1097/SIH.0000000000000189</pub-id><pub-id pub-id-type="medline">28704288</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaplonyi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bowles</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Nestel</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Understanding the impact of simulated patients on health care learners&#x2019; communication skills: a systematic review</article-title><source>Med Educ</source><year>2017</year><month>12</month><volume>51</volume><issue>12</issue><fpage>1209</fpage><lpage>1219</lpage><pub-id pub-id-type="doi">10.1111/medu.13387</pub-id><pub-id pub-id-type="medline">28833360</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Joyner</surname><given-names>B</given-names> </name><name name-style="western"><surname>Young</surname><given-names>L</given-names> </name></person-group><article-title>Teaching medical students using role play: twelve tips for successful role plays</article-title><source>Med Teach</source><year>2006</year><month>05</month><volume>28</volume><issue>3</issue><fpage>225</fpage><lpage>229</lpage><pub-id pub-id-type="doi">10.1080/01421590600711252</pub-id><pub-id pub-id-type="medline">16753719</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Perlman</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Anderson</surname><given-names>T</given-names> </name><name name-style="western"><surname>Foley</surname><given-names>VK</given-names> </name><name name-style="western"><surname>Mimnaugh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Safran</surname><given-names>JD</given-names> </name></person-group><article-title>The impact of alliance-focused and facilitative interpersonal relationship training on therapist skills: an RCT of brief training</article-title><source>Psychother Res</source><year>2020</year><month>09</month><volume>30</volume><issue>7</issue><fpage>871</fpage><lpage>884</lpage><pub-id pub-id-type="doi">10.1080/10503307.2020.1722862</pub-id><pub-id pub-id-type="medline">32028859</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fern&#x00E1;ndez-Alc&#x00E1;ntara</surname><given-names>M</given-names> </name><name name-style="western"><surname>Escribano</surname><given-names>S</given-names> </name><name name-style="western"><surname>Juli&#x00E1;-Sanchis</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Virtual simulation tools for communication skills training in health care professionals: literature review</article-title><source>JMIR Med Educ</source><year>2025</year><month>05</month><day>6</day><volume>11</volume><fpage>e63082</fpage><pub-id pub-id-type="doi">10.2196/63082</pub-id><pub-id pub-id-type="medline">40327882</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zeng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Qi</surname><given-names>W</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Embracing the future of medical education with large language model-based virtual patients: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>11</month><day>13</day><volume>27</volume><fpage>e79091</fpage><pub-id pub-id-type="doi">10.2196/79091</pub-id><pub-id pub-id-type="medline">41232097</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scholich</surname><given-names>T</given-names> </name><name name-style="western"><surname>Barr</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wiltsey Stirman</surname><given-names>S</given-names> </name><name name-style="western"><surname>Raj</surname><given-names>S</given-names> </name></person-group><article-title>A comparison of responses from human therapists and large language model-based chatbots to assess therapeutic communication: mixed methods study</article-title><source>JMIR Ment Health</source><year>2025</year><month>05</month><day>21</day><volume>12</volume><fpage>e69709</fpage><pub-id pub-id-type="doi">10.2196/69709</pub-id><pub-id pub-id-type="medline">40397927</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hettema</surname><given-names>J</given-names> </name><name name-style="western"><surname>Steele</surname><given-names>J</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>WR</given-names> </name></person-group><article-title>Motivational interviewing</article-title><source>Annu Rev Clin Psychol</source><year>2005</year><volume>1</volume><fpage>91</fpage><lpage>111</lpage><pub-id pub-id-type="doi">10.1146/annurev.clinpsy.1.102803.143833</pub-id><pub-id pub-id-type="medline">17716083</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Beck</surname><given-names>JS</given-names> </name></person-group><source>Cognitive Behavior Therapy: Basics and Beyond</source><year>2020</year><edition>3</edition><publisher-name>Guilford Publications</publisher-name><pub-id pub-id-type="other">9781462544196</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Moyers</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Manuel</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Ernst</surname><given-names>D</given-names> </name></person-group><article-title>Motivational Interviewing Treatment Integrity Coding Manual 4.2.1</article-title><year>2015</year><access-date>2026-07-28</access-date><publisher-name>University of New Mexico, Center on Alcoholism, Substance Abuse, and Addictions (CASAA)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://casaa.unm.edu/assets/docs/miti4_21.pdf">https://casaa.unm.edu/assets/docs/miti4_21.pdf</ext-link></comment></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Miller</surname><given-names>WR</given-names> </name><name name-style="western"><surname>Moyers</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Ernst</surname><given-names>D</given-names> </name><name name-style="western"><surname>Amrhein</surname><given-names>P</given-names> </name></person-group><article-title>Manual for the Motivational Interviewing Skill Code (MISC), Version 2.0</article-title><year>2003</year><access-date>2026-07-28</access-date><publisher-name>Center on Alcoholism, Substance Abuse and Addictions (CASAA), The University of New Mexico</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://digitalcommons.montclair.edu/cgi/viewcontent.cgi?article=1026&#x0026;context=psychology-facpubs">https://digitalcommons.montclair.edu/cgi/viewcontent.cgi?article=1026&#x0026;context=psychology-facpubs</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Young</surname><given-names>J</given-names> </name><name name-style="western"><surname>Beck</surname><given-names>AT</given-names> </name></person-group><article-title>Cognitive therapy scale rating manual</article-title><year>1980</year><access-date>2026-07-28</access-date><publisher-name>University of Pennsylvania</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://static1.squarespace.com/static/60f6fb2c51d9421daf31868c/t/6103d250c8ffd964aa4fe413/1627640400759/CTRS_Manual.pdf">https://static1.squarespace.com/static/60f6fb2c51d9421daf31868c/t/6103d250c8ffd964aa4fe413/1627640400759/CTRS_Manual.pdf</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Soma</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Kuo</surname><given-names>PB</given-names> </name><name name-style="western"><surname>Mehta</surname><given-names>M</given-names> </name><name name-style="western"><surname>Srikumar</surname><given-names>V</given-names> </name><name name-style="western"><surname>Imel</surname><given-names>ZE</given-names> </name><name name-style="western"><surname>Atkins</surname><given-names>DC</given-names> </name></person-group><article-title>Artificial intelligence to support human-provided mental health treatment</article-title><source>Annu Rev Clin Psychol</source><year>2026</year><month>05</month><volume>22</volume><issue>1</issue><fpage>505</fpage><lpage>531</lpage><pub-id pub-id-type="doi">10.1146/annurev-clinpsy-061724-075336</pub-id><pub-id pub-id-type="medline">41666031</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pellemans</surname><given-names>M</given-names> </name><name name-style="western"><surname>Salmi</surname><given-names>S</given-names> </name><name name-style="western"><surname>M&#x00E9;relle</surname><given-names>S</given-names> </name><name name-style="western"><surname>Janssen</surname><given-names>W</given-names> </name><name name-style="western"><surname>van der Mei</surname><given-names>R</given-names> </name></person-group><article-title>Automated behavioral coding to enhance the effectiveness of motivational interviewing in a chat-based suicide prevention helpline: secondary analysis of a clinical trial</article-title><source>J Med Internet Res</source><year>2024</year><month>08</month><day>1</day><volume>26</volume><fpage>e53562</fpage><pub-id pub-id-type="doi">10.2196/53562</pub-id><pub-id pub-id-type="medline">39088244</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lim</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jung</surname><given-names>YC</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>BH</given-names> </name></person-group><article-title>Evaluating motivational interview quality using large language models and hidden Markov models</article-title><source>BMC Psychiatry</source><year>2025</year><month>10</month><day>1</day><volume>25</volume><issue>1</issue><fpage>908</fpage><pub-id pub-id-type="doi">10.1186/s12888-025-07391-1</pub-id><pub-id pub-id-type="medline">41034852</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Current status of ChatGPT use in medical education: potentials, challenges, and strategies</article-title><source>J Med Internet Res</source><year>2024</year><month>08</month><day>28</day><volume>26</volume><fpage>e57896</fpage><pub-id pub-id-type="doi">10.2196/57896</pub-id><pub-id pub-id-type="medline">39196640</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abd-Alrazaq</surname><given-names>A</given-names> </name><name name-style="western"><surname>AlSaad</surname><given-names>R</given-names> </name><name name-style="western"><surname>Alhuwail</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Large language models in medical education: opportunities, challenges, and future directions</article-title><source>JMIR Med Educ</source><year>2023</year><month>06</month><day>1</day><volume>9</volume><fpage>e48291</fpage><pub-id pub-id-type="doi">10.2196/48291</pub-id><pub-id pub-id-type="medline">37261894</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Holderried</surname><given-names>F</given-names> </name><name name-style="western"><surname>Stegemann-Philipps</surname><given-names>C</given-names> </name><name name-style="western"><surname>Herrmann-Werner</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A language model-powered simulated patient with automated feedback for history taking: prospective study</article-title><source>JMIR Med Educ</source><year>2024</year><month>08</month><day>16</day><volume>10</volume><fpage>e59213</fpage><pub-id pub-id-type="doi">10.2196/59213</pub-id><pub-id pub-id-type="medline">39150749</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vrdoljak</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boban</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Vilovi&#x0107;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kumri&#x0107;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bo&#x017E;i&#x0107;</surname><given-names>J</given-names> </name></person-group><article-title>A review of large language models in medical education, clinical decision support, and healthcare administration</article-title><source>Healthcare (Basel)</source><year>2025</year><month>03</month><day>10</day><volume>13</volume><issue>6</issue><fpage>603</fpage><pub-id pub-id-type="doi">10.3390/healthcare13060603</pub-id><pub-id pub-id-type="medline">40150453</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barra</surname><given-names>FL</given-names> </name><name name-style="western"><surname>Rodella</surname><given-names>G</given-names> </name><name name-style="western"><surname>Costa</surname><given-names>A</given-names> </name><etal/></person-group><article-title>From prompt to platform: an agentic AI workflow for healthcare simulation scenario design</article-title><source>Adv Simul (Lond)</source><year>2025</year><month>05</month><day>16</day><volume>10</volume><issue>1</issue><fpage>29</fpage><pub-id pub-id-type="doi">10.1186/s41077-025-00357-z</pub-id><pub-id pub-id-type="medline">40380247</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>H</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>W</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Del Bue</surname><given-names>A</given-names> </name><name name-style="western"><surname>Canton</surname><given-names>C</given-names> </name><name name-style="western"><surname>Pont-Tuset</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tommasi</surname><given-names>T</given-names> </name></person-group><article-title>MEDCO: medical education copilots based on a multi-agent framework</article-title><source>Computer Vision &#x2013; ECCV 2024 Workshops Milan, Italy, September 29&#x2013;October 4, 2024, Proceedings, Part VIII</source><year>2024</year><publisher-name>Springer</publisher-name><fpage>119</fpage><lpage>135</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-91813-1_8</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sapkota</surname><given-names>R</given-names> </name><name name-style="western"><surname>Roumeliotis</surname><given-names>KI</given-names> </name><name name-style="western"><surname>Karkee</surname><given-names>M</given-names> </name></person-group><article-title>AI Agents vs. Agentic AI: a conceptual taxonomy, applications and challenges</article-title><source>Inf Fusion</source><year>2026</year><month>02</month><volume>126</volume><fpage>103599</fpage><pub-id pub-id-type="doi">10.1016/j.inffus.2025.103599</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Simulated patient systems powered by large language model-based AI agents offer potential for transforming medical education</article-title><source>Commun Med (Lond)</source><year>2025</year><month>12</month><day>19</day><volume>6</volume><issue>1</issue><fpage>27</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-01283-x</pub-id><pub-id pub-id-type="medline">41420084</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sha</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Practical and ethical challenges of large language models in education: a systematic scoping review</article-title><source>Brit J Educational Tech</source><year>2024</year><month>01</month><volume>55</volume><issue>1</issue><fpage>90</fpage><lpage>112</lpage><pub-id pub-id-type="doi">10.1111/bjet.13370</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Templin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Fort</surname><given-names>S</given-names> </name><name name-style="western"><surname>Padmanabham</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Framework for bias evaluation in large language models in healthcare settings</article-title><source>NPJ Digit Med</source><year>2025</year><month>07</month><day>7</day><volume>8</volume><issue>1</issue><fpage>414</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01786-w</pub-id><pub-id pub-id-type="medline">40624264</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stade</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Stirman</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Ungar</surname><given-names>LH</given-names> </name><etal/></person-group><article-title>Large language models could change the future of behavioral healthcare: a proposal for responsible development and evaluation</article-title><source>Npj Ment Health Res</source><year>2024</year><month>04</month><day>2</day><volume>3</volume><issue>1</issue><fpage>12</fpage><pub-id pub-id-type="doi">10.1038/s44184-024-00056-z</pub-id><pub-id pub-id-type="medline">38609507</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Yosef</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zisquit</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Klomek Brunstein</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bar</surname><given-names>K</given-names> </name><name name-style="western"><surname>Friedman</surname><given-names>D</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Ophir</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Desmet</surname><given-names>B</given-names> </name><name name-style="western"><surname>Prud&#x2019;hommeaux</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zirikly</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bedrick</surname><given-names>S</given-names> </name><name name-style="western"><surname>MacAvaney</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bar</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ireland</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ophir</surname><given-names>Y</given-names> </name></person-group><article-title>Assessing motivational interviewing sessions with AI-generated patient simulations</article-title><source>Proceedings of the 9th Workshop on Computational Linguistics and Clinical Psychology (CLPsych 2024)</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.clpsych-1.1</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="book"><source>Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition</source><year>2013</year><access-date>2026-07-31</access-date><publisher-name>American Psychiatric Association</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://psychiatryonline.org/doi/book/10.1176/appi.books.9780890425596">https://psychiatryonline.org/doi/book/10.1176/appi.books.9780890425596</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Balloccu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Anno-MI: a dataset of expert-annotated counselling dialogues</article-title><source>ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source><year>2022</year><publisher-name>IEEE</publisher-name><fpage>6177</fpage><lpage>6181</lpage><pub-id pub-id-type="doi">10.1109/ICASSP43922.2022.9746035</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vasey</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nagendran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Campbell</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Reporting guideline for the early-stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title><source>Nat Med</source><year>2022</year><month>05</month><volume>28</volume><issue>5</issue><fpage>924</fpage><lpage>933</lpage><pub-id pub-id-type="doi">10.1038/s41591-022-01772-9</pub-id><pub-id pub-id-type="medline">35585198</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="report"><person-group person-group-type="author"><collab>Canadian Institutes of Health Research; Natural Sciences and Engineering Research Council of Canada; Social Sciences and Humanities Research Council of Canada (Tri-Council agencies)</collab></person-group><article-title>Tri-council policy statement: ethical conduct for research involving humans</article-title><year>2022</year><access-date>2026-07-28</access-date><publisher-name>Interagency Secretariat on Research Ethics</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://ethics.gc.ca/eng/documents/tcps2-2022-en.pdf">https://ethics.gc.ca/eng/documents/tcps2-2022-en.pdf</ext-link></comment></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Balloccu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>V</given-names> </name><name name-style="western"><surname>Helaoui</surname><given-names>R</given-names> </name><name name-style="western"><surname>Reforgiato Recupero</surname><given-names>D</given-names> </name><name name-style="western"><surname>Riboni</surname><given-names>D</given-names> </name></person-group><article-title>Creation, analysis and evaluation of AnnoMI, a dataset of expert-annotated counselling dialogues</article-title><source>Future Internet</source><year>2023</year><volume>15</volume><issue>3</issue><fpage>110</fpage><pub-id pub-id-type="doi">10.3390/fi15030110</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>GB</given-names> </name><name name-style="western"><surname>Chiu</surname><given-names>AM</given-names> </name></person-group><article-title>Assessment and feedback methods in competency-based medical education</article-title><source>Ann Allergy Asthma Immunol</source><year>2022</year><month>03</month><volume>128</volume><issue>3</issue><fpage>256</fpage><lpage>262</lpage><pub-id pub-id-type="doi">10.1016/j.anai.2021.12.010</pub-id><pub-id pub-id-type="medline">34929390</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A survey on LLM-as-a-judge</article-title><source>Innovation (Camb)</source><year>2026</year><month>06</month><day>1</day><volume>7</volume><issue>6</issue><fpage>101253</fpage><pub-id pub-id-type="doi">10.1016/j.xinn.2025.101253</pub-id><pub-id pub-id-type="medline">42254963</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tam</surname><given-names>TYC</given-names> </name><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kapoor</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>28</day><volume>7</volume><issue>1</issue><fpage>258</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id><pub-id pub-id-type="medline">39333376</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cook</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Hatala</surname><given-names>R</given-names> </name></person-group><article-title>Validation of educational assessments: a primer for simulation and beyond</article-title><source>Adv Simul (Lond)</source><year>2016</year><volume>1</volume><fpage>31</fpage><pub-id pub-id-type="doi">10.1186/s41077-016-0033-y</pub-id><pub-id pub-id-type="medline">29450000</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>T</given-names> </name><name name-style="western"><surname>Hou</surname><given-names>B</given-names> </name><etal/></person-group><article-title>MentalChat16K: a benchmark dataset for conversational mental health assistance</article-title><source>KDD</source><year>2025</year><month>08</month><volume>2025</volume><fpage>5367</fpage><lpage>5378</lpage><pub-id pub-id-type="doi">10.1145/3711896.3737393</pub-id><pub-id pub-id-type="medline">41098434</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>WL</given-names> </name><name name-style="western"><surname>Sheng</surname><given-names>Y</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Oh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Globerson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Saenko</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hardt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Levine</surname><given-names>S</given-names> </name></person-group><article-title>Judging LLM-as-a-judge with MT-bench and chatbot arena</article-title><source>NIPS &#x2019;23: Proceedings of the 37th International Conference on Neural Information Processing Systems</source><year>2023</year><access-date>2026-07-28</access-date><publisher-name>Curran Associates Inc</publisher-name><fpage>46595</fpage><lpage>46623</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.5555/3666122.3668142">https://dl.acm.org/doi/10.5555/3666122.3668142</ext-link></comment></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Brin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Barash</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Large language models and empathy: systematic review</article-title><source>J Med Internet Res</source><year>2024</year><month>12</month><day>11</day><volume>26</volume><fpage>e52597</fpage><pub-id pub-id-type="doi">10.2196/52597</pub-id><pub-id pub-id-type="medline">39661968</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cook</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Hatala</surname><given-names>R</given-names> </name><name name-style="western"><surname>Brydges</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Technology-enhanced simulation for health professions education: a systematic review and meta-analysis</article-title><source>JAMA</source><year>2011</year><month>09</month><day>7</day><volume>306</volume><issue>9</issue><fpage>978</fpage><lpage>988</lpage><pub-id pub-id-type="doi">10.1001/jama.2011.1234</pub-id><pub-id pub-id-type="medline">21900138</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Owen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hilsenroth</surname><given-names>MJ</given-names> </name></person-group><article-title>Treatment adherence: the importance of therapist flexibility in relation to therapy outcomes</article-title><source>J Couns Psychol</source><year>2014</year><month>04</month><volume>61</volume><issue>2</issue><fpage>280</fpage><lpage>288</lpage><pub-id pub-id-type="doi">10.1037/a0035753</pub-id><pub-id pub-id-type="medline">24635591</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McHugh</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Murray</surname><given-names>HW</given-names> </name><name name-style="western"><surname>Barlow</surname><given-names>DH</given-names> </name></person-group><article-title>Balancing fidelity and adaptation in the dissemination of empirically-supported treatments: the promise of transdiagnostic interventions</article-title><source>Behav Res Ther</source><year>2009</year><month>11</month><volume>47</volume><issue>11</issue><fpage>946</fpage><lpage>953</lpage><pub-id pub-id-type="doi">10.1016/j.brat.2009.07.005</pub-id><pub-id pub-id-type="medline">19643395</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fl&#x00FC;ckiger</surname><given-names>C</given-names> </name><name name-style="western"><surname>Del Re</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Wampold</surname><given-names>BE</given-names> </name><name name-style="western"><surname>Horvath</surname><given-names>AO</given-names> </name></person-group><article-title>The alliance in adult psychotherapy: a meta-analytic synthesis</article-title><source>Psychotherapy (Chic)</source><year>2018</year><month>12</month><volume>55</volume><issue>4</issue><fpage>316</fpage><lpage>340</lpage><pub-id pub-id-type="doi">10.1037/pst0000172</pub-id><pub-id pub-id-type="medline">29792475</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eubanks</surname><given-names>CF</given-names> </name><name name-style="western"><surname>Muran</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Safran</surname><given-names>JD</given-names> </name></person-group><article-title>Alliance rupture repair: a meta-analysis</article-title><source>Psychotherapy (Chic)</source><year>2018</year><month>12</month><volume>55</volume><issue>4</issue><fpage>508</fpage><lpage>519</lpage><pub-id pub-id-type="doi">10.1037/pst0000185</pub-id><pub-id pub-id-type="medline">30335462</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elhilali</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ngo</surname><given-names>ASH</given-names> </name><name name-style="western"><surname>Reichenpfader</surname><given-names>D</given-names> </name><name name-style="western"><surname>Denecke</surname><given-names>K</given-names> </name></person-group><article-title>Large language model-based patient simulation to foster communication skills in health care professionals: user-centered development and usability study</article-title><source>JMIR Med Educ</source><year>2025</year><month>12</month><day>12</day><volume>11</volume><fpage>e81271</fpage><pub-id pub-id-type="doi">10.2196/81271</pub-id><pub-id pub-id-type="medline">41385781</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Deng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Heng</surname><given-names>Y</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Ku</surname><given-names>LW</given-names> </name><name name-style="western"><surname>Martins</surname><given-names>A</given-names> </name><name name-style="western"><surname>Srikumar</surname><given-names>V</given-names> </name></person-group><article-title>Unveiling the spectrum of data contamination in language model: a survey from detection to remediation</article-title><source>Findings of the Association for Computational Linguistics: ACL 2024</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>16078</fpage><lpage>16092</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.951</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thygesen</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Farrington</surname><given-names>J</given-names> </name><name name-style="western"><surname>Keen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Large language models for mental health applications: systematic review</article-title><source>JMIR Ment Health</source><year>2024</year><month>10</month><day>18</day><volume>11</volume><fpage>e57400</fpage><pub-id pub-id-type="doi">10.2196/57400</pub-id><pub-id pub-id-type="medline">39423368</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Na</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A scoping review of large language models for generative tasks in mental health care</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>30</day><volume>8</volume><issue>1</issue><fpage>230</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01611-4</pub-id><pub-id pub-id-type="medline">40307331</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bjaastad</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Lillevoll</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hoffart</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The effect of behavioral rehearsal in the training of clinical psychology students in cognitive therapy for social anxiety disorder: a randomized controlled trial</article-title><source>Behav Res Ther</source><year>2025</year><month>10</month><volume>193</volume><fpage>104849</fpage><pub-id pub-id-type="doi">10.1016/j.brat.2025.104849</pub-id><pub-id pub-id-type="medline">40907373</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miller</surname><given-names>WR</given-names> </name><name name-style="western"><surname>Yahne</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Moyers</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Martinez</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pirritano</surname><given-names>M</given-names> </name></person-group><article-title>A randomized trial of methods to help clinicians learn motivational interviewing</article-title><source>J Consult Clin Psychol</source><year>2004</year><month>12</month><volume>72</volume><issue>6</issue><fpage>1050</fpage><lpage>1062</lpage><pub-id pub-id-type="doi">10.1037/0022-006X.72.6.1050</pub-id><pub-id pub-id-type="medline">15612851</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Imel</surname><given-names>ZE</given-names> </name><name name-style="western"><surname>Baldwin</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Baer</surname><given-names>JS</given-names> </name><etal/></person-group><article-title>Evaluating therapist adherence in motivational interviewing by comparing performance with standardized and real patients</article-title><source>J Consult Clin Psychol</source><year>2014</year><month>06</month><volume>82</volume><issue>3</issue><fpage>472</fpage><lpage>481</lpage><pub-id pub-id-type="doi">10.1037/a0036158</pub-id><pub-id pub-id-type="medline">24588405</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bubeck</surname><given-names>S</given-names> </name><name name-style="western"><surname>Petro</surname><given-names>J</given-names> </name></person-group><article-title>Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine</article-title><source>N Engl J Med</source><year>2023</year><month>03</month><day>30</day><volume>388</volume><issue>13</issue><fpage>1233</fpage><lpage>1239</lpage><pub-id pub-id-type="doi">10.1056/NEJMsr2214184</pub-id><pub-id pub-id-type="medline">36988602</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chelli</surname><given-names>M</given-names> </name><name name-style="western"><surname>Descamps</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lavou&#x00E9;</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Hallucination rates and reference accuracy of ChatGPT and Bard for systematic reviews: comparative analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>05</month><day>22</day><volume>26</volume><fpage>e53164</fpage><pub-id pub-id-type="doi">10.2196/53164</pub-id><pub-id pub-id-type="medline">38776130</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Schaekermann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Palepu</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Towards conversational diagnostic artificial intelligence</article-title><source>Nature</source><year>2025</year><month>06</month><volume>642</volume><issue>8067</issue><fpage>442</fpage><lpage>450</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08866-7</pub-id><pub-id pub-id-type="medline">40205050</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Del Giacco</surname><given-names>L</given-names> </name><name name-style="western"><surname>Anguera</surname><given-names>MT</given-names> </name><name name-style="western"><surname>Salcuni</surname><given-names>S</given-names> </name></person-group><article-title>The action of verbal and non-verbal communication in the therapeutic alliance construction: a mixed methods approach to assess the initial interactions with depressed patients</article-title><source>Front Psychol</source><year>2020</year><volume>11</volume><fpage>234</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2020.00234</pub-id><pub-id pub-id-type="medline">32153459</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Spiliopoulou</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Play favorites: a statistical method to measure self-bias in LLM-as-a-judge</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 8, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.06709</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Greene</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kechadi</surname><given-names>MT</given-names> </name></person-group><article-title>Benchmark data contamination of large language models: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 6, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2406.04244</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="web"><article-title>amin-kamaleddin/MI-and-CBT-agent</article-title><source>GitHub</source><access-date>2026-07-28</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/amin-kamaleddin/MI-and-CBT-Agent">https://github.com/amin-kamaleddin/MI-and-CBT-Agent</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary figure and tables.</p><media xlink:href="mededu_v12i1e92964_app1.docx" xlink:title="DOCX File, 10224 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 1</label><p>DECIDE-AI checklist.</p><media xlink:href="mededu_v12i1e92964_app2.docx" xlink:title="DOCX File, 2170 KB"/></supplementary-material></app-group></back></article>