<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id><journal-id journal-id-type="publisher-id">mededu</journal-id><journal-id journal-id-type="index">20</journal-id><journal-title>JMIR Medical Education</journal-title><abbrev-journal-title>JMIR Med Educ</abbrev-journal-title><issn pub-type="epub">2369-3762</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v12i1e91809</article-id><article-id pub-id-type="doi">10.2196/91809</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Multimodal Large Language Models for Dental Chart Image Interpretation: Cross-Sectional Benchmarking Study With Students and Clinicians</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Cho</surname><given-names>Ah-Young</given-names></name><degrees>DDS, MSD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Eo</surname><given-names>Soo-Heang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jeon</surname><given-names>Mi-Jeong</given-names></name><degrees>DDS, MSD, PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kim</surname><given-names>Jae-Hoon</given-names></name><degrees>DDS, MSD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ihm</surname><given-names>Jungjoon</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Seo</surname><given-names>Deog-Gyu</given-names></name><degrees>DDS, MSD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Conservative Dentistry and Dental Research Institute, School of Dentistry, Seoul National University</institution><addr-line>101 Daehakno, Jongno-Gu</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff2"><institution>Graduate School, Department of Urban Big Data Convergence, University of Seoul</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff3"><institution>AI Agent Team, CryptoLab Inc.</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff4"><institution>Department of Conservative Dentistry and Oral Science Research Center, Yonsei University College of Dentistry</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff5"><institution>Department of Dental Education, Dental and Life Science Institute, School of Dentistry, Pusan National University and Dental Research Institute</institution><addr-line>Busan</addr-line><country>Republic of Korea</country></aff><aff id="aff6"><institution>Department of Dental Education and Dental Research Institute, School of Dentistry, Seoul National University</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Kanzow</surname><given-names>Philipp</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Pang</surname><given-names>MengWei</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Dhawan</surname><given-names>Pankaj</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Watanabe</surname><given-names>Plauto Christopher Aranha</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Deog-Gyu Seo, DDS, MSD, PhD, Department of Conservative Dentistry and Dental Research Institute, School of Dentistry, Seoul National University, 101 Daehakno, Jongno-Gu, Seoul, 03080, Republic of Korea, 82 2-6256-3182; <email>dgseo@snu.ac.kr</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>9</month><year>2026</year></pub-date><volume>12</volume><elocation-id>e91809</elocation-id><history><date date-type="received"><day>25</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>14</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>17</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Ah-Young Cho, Soo-Heang Eo, Mi-Jeong Jeon, Jae-Hoon Kim, Jungjoon Ihm, Deog-Gyu Seo. Originally published in JMIR Medical Education (<ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org">https://mededu.jmir.org</ext-link>), 4.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org/">https://mededu.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mededu.jmir.org/2026/1/e91809"/><abstract><sec><title>Background</title><p>Entering clinical training, dental students must learn to read tooth-centered electronic dental records, but limited teaching time in patient-centered clinics can leave gaps in chart-reading literacy. Multimodal large language models (LLMs) that process dental record images may offer scalable educational support, yet their performance has not been benchmarked.</p></sec><sec><title>Objective</title><p>This study evaluated multimodal ChatGPT models on dental chart image interpretation and compared the best-performing model with dental students and residents.</p></sec><sec sec-type="methods"><title>Methods</title><p>A retrospective, cross-sectional benchmark study used deidentified dental chart images from 15 patients at Seoul National University Dental Hospital (2017&#x2010;2025). Charts containing Korean and English text were captured as sequential screenshots (154 images). For each patient, 24 Korean-language questions (360 total) spanned four categories: Type A, general factual retrieval; Type B, tooth- or procedure-specific retrieval; Type C, interpretation requiring multientry synthesis; and Type D, absent-information questions assessing abstention. Nine multimodal ChatGPT models (available August 2, 2025&#x2010;August 9, 2025) were evaluated under standardized conditions. Outputs were scored against a gold standard using 7 metrics, with sentence bidirectional encoder representations from transformers (SBERT) similarity prespecified as the primary semantic measure. Human baselines included 2 third-year students and 2 first-year residents. Groups were compared with Kruskal-Wallis tests and Dunn post hoc analyses. Gold-standard reliability was assessed by independent senior-expert review and chance-corrected agreement (Gwet AC1) among clinical reference raters, and the model ranking was confirmed by content-based clinical accuracy analysis.</p></sec><sec sec-type="results"><title>Results</title><p>Across 360 items, GPT-5 Thinking achieved the highest median SBERT similarity (0.900, IQR 0.525&#x2010;1.000), followed by GPT-5 Pro (0.861, IQR 0.501&#x2010;1.000) and OpenAI o3 (0.831, IQR 0.489&#x2010;1.000), with significant overall group differences (<italic>P</italic>&#x003C;.001). Student 1 did not differ significantly from GPT-5 Thinking across Types A-D (all <italic>P</italic>&#x2265;.16), whereas Student 2 differed only on Type D (<italic>P</italic>=.002; Cliff &#x03B4;=&#x2212;0.27). Resident 1 scored higher than GPT-5 Thinking on Types A (<italic>P</italic>=.01) and C (<italic>P</italic>=.04) but lower on Type D (<italic>P</italic>&#x003C;.001; &#x03B4;=&#x2212;0.35), whereas Resident 2 scored higher on Types A (<italic>P</italic>=.03), B (<italic>P</italic>=.02), and C (<italic>P</italic>=.04) and did not differ on Type D (<italic>P</italic>=.23). Type C tasks showed compressed SBERT distributions and low exact-match rates, indicating persistent difficulty in synthesis. The clinical reference rater agreement was high (Gwet AC1=0.99), and the content-based clinical-accuracy ranking was closely aligned with the SBERT ranking (Spearman &#x03C1;=0.90; <italic>P</italic>=.001).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Statistically nonsignificant differences were observed between GPT-5 Thinking and dental students for most question-type contrasts, with Student 2 differing only on absent-information items. Compared with first-year residents, GPT-5 Thinking remained lower on several Type A-C contrasts, particularly interpretive Type C and one tooth- or procedure-specific Type B comparison; Type D contrasts require cautious interpretation because all groups had ceiling medians. Within this single-center benchmark, multimodal LLMs may have potential as supervised educational tools for chart-reading practice and verification, rather than as replacements for clinical expertise, pending external validation across institutions, specialties, and record systems.</p></sec></abstract><kwd-group><kwd>dental records</kwd><kwd>education, dental</kwd><kwd>students, dental</kwd><kwd>clinical competence</kwd><kwd>large language models</kwd><kwd>generative AI</kwd><kwd>ChatGPT</kwd><kwd>multimodal ChatGPT</kwd><kwd>benchmarking</kwd><kwd>natural language processing</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>At the preclinical-to-clinical transition, when learners first enter the clinical environment, dental students must develop the ability to read and interpret dental charts, also referred to as dental records, including locating relevant information, reconstructing treatment timelines, and distinguishing tooth-specific details from patient-level records [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. For students new to clinical practice, this process can be challenging and may add cognitive load in already demanding learning environments, particularly within patient-centered clinics where clinical requirements limit protected teaching time [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. Although dental charts are foundational records of a patient&#x2019;s history and guide subsequent treatment, limited clinical exposure can lead to chart misinterpretation, increasing the downstream risk of treatment errors [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Because this transition is formative, curricula should minimize gaps by providing explicit early instruction and structured guidance to support competence development. However, current dental curricula often prioritize hands-on procedural training, leaving the cognitive skills required for interpreting clinical records largely implicit [<xref ref-type="bibr" rid="ref4">4</xref>]. The volume and complexity of chart data, coupled with the clinical priorities of patient-centered care, further constrain opportunities for timely teaching and feedback. Ideally, faculty would regularly assess students&#x2019; chart-reading skills and provide targeted coaching; in practice, patient-care responsibilities limit real-time instruction, and scalable assessment of comprehension is rarely feasible. Consequently, developing &#x201C;chart-reading literacy&#x201D;&#x2014;the ability to locate, interpret, and synthesize information from complex dental charts&#x2014;represents a core educational objective and a prerequisite for clinical reasoning and procedural competence in dentistry [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>Against this backdrop, large language models (LLMs) have emerged as potential educational aids. LLMs can retrieve, organize, and synthesize information into coherent, task-focused explanations, and recent advances in training and alignment have substantially improved their accuracy and robustness [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. ChatGPT, in particular, is a widely renowned LLM encompassing multiple releases, including reasoning-focused models and the GPT-4 series; most recently, the GPT-5 family (GPT-5, GPT-5 Thinking, and GPT-5 Pro) has expanded multimodal capabilities and strengthened reliability and reasoning. Learners across disciplines, including dental students, increasingly use ChatGPT to seek information and clarify unfamiliar topics [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Its chat-based interface supports AI-guided learning by enabling users to pose questions, test interpretations, and iteratively refine understanding through interactive dialogue. For students navigating the preclinical-to-clinical transition, ChatGPT may facilitate verification of chart interpretations, provide immediate clarification, and model structured reasoning in real time. Consequently, generative AI-guided education has been proposed as an emerging approach in dental education, positioning LLMs as adaptive instructional partners rather than passive information sources [<xref ref-type="bibr" rid="ref11">11</xref>]. At the same time, persistent limitations&#x2014;including hallucinations, sensitivity to prompt formulation, and variable performance across task types&#x2014;highlight the continued need for supervision and verification [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>Dental electronic records present distinct interpretive challenges that distinguish them from general electronic medical records (EMRs). Whereas most EMRs are organized around patient-level narrative entries, dental records are organized at the level of individual teeth and integrate spatial elements such as odontograms, tooth-numbering grids, and procedure maps [<xref ref-type="bibr" rid="ref14">14</xref>]. Relevant information is frequently distributed across multiple pages and time points so that even straightforward tasks, such as reconstructing the treatment history of a single tooth, require learners to integrate spatial cues with textual entries [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Critically, these spatial cues cannot be faithfully preserved when dental records are linearized into text alone, and text-only approaches therefore fail to capture the tooth-level context central to accurate interpretation of dental records [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. This challenge is compounded by substantial heterogeneity and documentation gaps in dental EMRs, which prior work in dental informatics has highlighted in calls for dental-specific templates and standardization [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. These characteristics indicate that meaningful evaluation of generative AI for dental chart reading requires multimodal inputs in which models reason directly over images of dental charts.</p><p>Despite this potential, few rigorous assessments have examined how generative AI models perform on dental chart images, particularly in educational contexts. Prior research in dental informatics has largely focused on coding systems, data quality, and secondary data use, with limited investigations into whether LLMs can accurately retrieve tooth-level information, reason across dispersed chart entries, and appropriately abstain when information is absent [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. This gap is consequential because well-characterized tools could reduce barriers to chart comprehension, support deliberate practice in clinical reasoning, and free instructional time for supervised skill acquisition, whereas poorly characterized tools risk amplifying errors and fostering false confidence. If ChatGPT&#x2019;s performance in educational settings is systematically validated, it may provide students navigating the preclinical-to-clinical transition with a scalable means to practice chart reading and verify their interpretations, even in the absence of formalized chart-reading instruction [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>This study evaluated multimodal ChatGPT models using dental chart images as inputs. The official models evaluated were OpenAI o3, GPT-4.5, GPT-4.1, GPT-4.1 mini, OpenAI o4-mini, GPT-4o, GPT-5, GPT-5 Thinking, and GPT-5 Pro. Models were benchmarked using semantic and exact-match evaluation metrics across 4 question categories, including general factual retrieval, tooth- or procedure-specific retrieval, interpretive reasoning, and absent-information (unanswerable) tasks. The highest-performing model was subsequently compared with dental students at the preclinical-to-clinical transition and with resident clinicians to assess relative performance in an educational context. This study aimed to benchmark the chart-image interpretation performance of multimodal ChatGPT models across these 4 task categories and to compare the best-performing model with dental students at the preclinical-to-clinical transition and with qualified clinicians in order to characterize where such models responsibly support chart-reading education.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Overview</title><p>A retrospective, cross-sectional evaluation was conducted using deidentified dental chart images under standardized testing conditions. The reporting of this study followed the STROBE (Strengthening the Reporting of Observational Studies in Epidemiology) checklist (<xref ref-type="supplementary-material" rid="app4">Checklist 1</xref>). The primary objective was to assess multimodal ChatGPT performance across three education-relevant competencies: (1) factual retrieval from dental charts, including general and tooth- or procedure-specific information, (2) inferential reasoning across dispersed chart entries, and (3) appropriate handling of missing or absent information. Analyses were prespecified and reported overall, by model, and by question type. The study workflow is summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flowchart of study design and evaluation process. BERT: bidirectional encoder representations from transformers; BLEU: bilingual evaluation understudy; EM: exact match; ROUGE-L: recall-oriented understudy for gisting evaluation-longest common subsequence; SBERT: sentence bidirectional encoder representations from transformers.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e91809_fig01.png"/></fig></sec><sec id="s2-2"><title>Data Source and Case Selection</title><p>Dental charts from 15 patients were sampled from the Department of Conservative Dentistry at Seoul National University Dental Hospital (SNUDH), covering the period from January 1, 2017, to June 30, 2025. The charts contained both Korean and English text. <xref ref-type="fig" rid="figure2">Figure 2</xref> presents an example of a captured dental chart image, including the patient&#x2019;s chief complaint, past medical and dental histories, current dental problems (present illness) with associated odontograms, clinical diagnoses, treatment plans, treatments performed, and free-text clinical notes. The case mix comprised restorative (n=6), endodontic (n=5), surgical (n=2), and mixed or complex cases (n=2). Inclusion criteria were as follows: (1) documentation of treatment for at least one tooth, (2) the presence of chart elements verifiable from images (eg, odontograms, procedure notes, and visit dates), and (3) complete deidentification prior to data export. Charts containing any personally identifiable information were excluded. Where feasible, cases were selected from multiple providers to reduce operator-specific variability.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Example structure of an image-captured dental chart format. CAA: chronic apical abscess; EPT: electric pulp test; EXT: extraction; GP: gutta percha; L/A: local anesthesia; PA: periapical; PPD: probing pocket depth; WNL: within normal limits.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e91809_fig02.png"/></fig></sec><sec id="s2-3"><title>Image Acquisition and Preprocessing</title><p>Dental charts were captured as sequential screenshots with a resolution of 717&#x00D7;742 pixels. Consecutive images overlapped by at least one line of text to prevent truncation at page boundaries. The number of images per patient ranged from 3 to 26 (mean 10.27, SD 6.54; total 154 images). Because the testing interface allowed a maximum of 10 images per message, larger cases exceeding this limit were uploaded in consecutive batches within the same conversation thread to preserve page order and context. All images were manually reviewed to ensure legibility and complete deidentification prior to use.</p></sec><sec id="s2-4"><title>Question Dataset and References</title><p>For each patient, 24 individualized questions in Korean were developed, resulting in a total of 360 questions. Questions were evenly distributed across 4 categories (<xref ref-type="table" rid="table1">Table 1</xref>):</p><list list-type="bullet"><list-item><p>Type A, general (6 questions): explicit chart facts, such as dates and provider names. Three items were common across all cases: first-visit date, total number of visits, and primary provider name.</p></list-item><list-item><p>Type B, dental (6 questions): tooth- or procedure-specific clinical details, for example, the date of a specific restoration or the presence of a final crown.</p></list-item><list-item><p>Type C, interpretation (6 questions): inference questions requiring synthesis of evidence across multiple chart entries, such as the rationale for a procedure.</p></list-item><list-item><p>Type D, absent information (6 questions): intentionally unanswerable questions designed to test appropriate abstention.</p></list-item></list><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Question types for dental chart evaluation: examples and rationale for large language model (LLM) assessment.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Category</td><td align="left" valign="bottom">Number of questions</td><td align="left" valign="bottom">Examples</td><td align="left" valign="bottom">Rationale for LLM<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> evaluation</td></tr></thead><tbody><tr><td align="left" valign="top">General</td><td align="left" valign="top">6</td><td align="left" valign="top">When is the first visit to the clinic?</td><td align="left" valign="top">Evaluation of basic recognition of simple information. Suitable for assessing OCR<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> accuracy and fundamental contextual understanding.</td></tr><tr><td align="left" valign="top">Dental</td><td align="left" valign="top">6</td><td align="left" valign="top">When was tooth number 16 treated for a cavity?</td><td align="left" valign="top">Recognition of technical dental terms. Useful for testing the ability to extract relevant details and apply subject-specific understanding.</td></tr><tr><td align="left" valign="top">Interpretation</td><td align="left" valign="top">6</td><td align="left" valign="top">For what purpose was the MTA<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> used for the treatment?</td><td align="left" valign="top">Assesses the ability to synthesize information from multiple sources and make informed inferences. Enables evaluation of integrated reasoning skills.</td></tr><tr><td align="left" valign="top">Absent information</td><td align="left" valign="top">6</td><td align="left" valign="top">What was the length of MB2<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> canal in tooth number 11? (despite the absence of MB2 canal)</td><td align="left" valign="top">Checks whether the system avoids generating incorrect information not present in the clinical chart. Essential for evaluating hallucination control capabilities.</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table1fn2"><p><sup>b</sup>OCR: optical character recognition.</p></fn><fn id="table1fn3"><p><sup>c</sup>MTA: mineral trioxide aggregate.</p></fn><fn id="table1fn4"><p><sup>d</sup>MB2: mesiobuccal 2.</p></fn></table-wrap-foot></table-wrap><p>The 4 question categories integrate Bloom taxonomy and established natural language processing (NLP) evaluation paradigms [<xref ref-type="bibr" rid="ref18">18</xref>]. Types A and B assess foundational information retrieval via single-hop queries, distinguishing baseline reading comprehension (Type A) from dental-specific domain expertise (Type B). Type C evaluates higher-order clinical reasoning, requiring multihop inference across dispersed chart entries. Type D adapts the unanswerable questions paradigm from robust NLP benchmarks (eg, SQuAD 2.0 and emrQA [<xref ref-type="bibr" rid="ref19">19</xref>]) to assess hallucination control and appropriate abstention, a critical safety prerequisite.</p><p>Questions were manually formulated by a single clinical researcher (AYC) to accurately reflect the unique clinical context of each patient case. Methodological consistency was ensured by anchoring all generated questions to the predefined conceptual framework (Types A, B, C, and D). As the primary objective of this study was to compare relative model performance, all evaluated models were presented with the identical set of case-specific questions, ensuring a fair and consistent comparative benchmark. The questions were authored by a single clinical researcher (AYC) and were not independently reviewed for wording prior to evaluation; however, the associated gold-standard answers were subsequently verified by an independent senior specialist.</p><p>Gold-standard answers were provided by a licensed dentist and full professor in the Department of Conservative Dentistry at SNUDH, with over 20 years of clinical experience. Two first-year residents from the same department contributed additional comparison answers. In this study, these residents were operationally defined as the &#x201C;resident clinician group.&#x201D; This terminology is justified as these individuals hold valid dental licenses and possess approximately 3-4 years of total clinical experience when accounting for their undergraduate clinical training. Consequently, they are considered capable of independent patient management and provided a formally qualified clinical baseline. For the student baseline, 2 third-year dental students at the preclinical-to-clinical transition (2 years of preclinical followed by 2 years of clinical curriculum) produced independent answer sets. This level was chosen to reflect students entering initial clinical exposure with limited prior experience in chart interpretation. All human raters completed the same set of 24 individualized questions per patient independently and without access to model outputs.</p></sec><sec id="s2-5"><title>Model Conditions and Inference Procedure</title><p>Nine ChatGPT models available from August 2-9, 2025, in Korea Standard Time via the ChatGPT (OpenAI) user interface were evaluated, namely OpenAI o3, GPT-4.5, GPT-4.1, GPT-4.1 mini, OpenAI o4-mini, GPT-4o, GPT-5, GPT-5 Thinking, and GPT-5 Pro. Evaluations were performed under a ChatGPT Pro subscription, the highest consumer tier available at the time. The study included all image-capable models selectable in the consumer interface during this window rather than a performance-based subset. Model inclusion was therefore governed by interface availability under this subscription tier; GPT-5 Pro, in particular, was accessible only at the Pro tier. For each patient-model pair, a new conversation was initiated to prevent cross-case contamination. All chart images for the patient, including continuation batches, and the corresponding 24 questions were presented together, accompanied by standardized Korean instructions directing the model to review all images and answer the questions. To ensure methodological transparency and reproducibility, the direct links to the complete conversational logs with the LLMs, including all exact prompts and generated responses, are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Conversation history was reset between models and between patients. Model temperature and other settings were left at platform defaults. All evaluated models support text-and-image inputs in ChatGPT; however, formal optical character recognition (OCR) accuracy metrics for these interfaces are not published. Context limits vary by model (approximately GPT-4.1 up to ~1 million tokens via API, GPT-5 up to ~400,000 tokens via API, and ~256,000 in ChatGPT). Comparative specifications of each model, including type, multimodal data support, context-window size, visual input handling, distinguishing features, and release date, are summarized in <xref ref-type="table" rid="table2">Table 2</xref> [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. The consumer interface was used to reflect how students and clinicians actually access these models, prioritizing ecological validity. The trade-off is that platform-level factors&#x2014;proprietary system prompts, memory-state handling, rate-limiting behavior, and per-query routing decisions, including GPT-5 smart-routing&#x2014;are not exposed and could not be independently verified or held under experimental control.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Characteristics of ChatGPT models, including type, multimodal data support, context length, image-processing capability, key features, and release date.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">AI model</td><td align="left" valign="bottom">Type</td><td align="left" valign="bottom">Data support</td><td align="left" valign="bottom">Context length</td><td align="left" valign="bottom">Image understanding performance</td><td align="left" valign="bottom">Key features</td><td align="left" valign="bottom">Release date</td></tr></thead><tbody><tr><td align="left" valign="top">OpenAI o3</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Text+image</td><td align="left" valign="top">Not disclosed</td><td align="left" valign="top">Supports visual input; no explicit OCR<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> evaluation</td><td align="left" valign="top">Optimized for structured scientific and coding reasoning, cost-efficient reasoning engine</td><td align="left" valign="top">April 16, 2025</td></tr><tr><td align="left" valign="top">GPT-4.5</td><td align="left" valign="top">GPT</td><td align="left" valign="top">Text+image</td><td align="left" valign="top">128,000 tokens</td><td align="left" valign="top">Supports visual input; no explicit OCR evaluation</td><td align="left" valign="top">Enhanced natural, intuitive dialogue, and aesthetic creativity; reduced hallucination and emotional IQ boost</td><td align="left" valign="top">February 27, 2025</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">GPT</td><td align="left" valign="top">Text+image</td><td align="left" valign="top">1 million tokens</td><td align="left" valign="top">Supports visual input; no explicit OCR evaluation</td><td align="left" valign="top">Large context for vision tasks</td><td align="left" valign="top">April 14, 2025</td></tr><tr><td align="left" valign="top">GPT-4.1 mini</td><td align="left" valign="top">GPT</td><td align="left" valign="top">Text+image</td><td align="left" valign="top">1 million tokens</td><td align="left" valign="top">Same as above; no specific OCR info</td><td align="left" valign="top">Scaled-down version of GPT-4.1; cost-efficient for similar tasks</td><td align="left" valign="top">April 14, 2025</td></tr><tr><td align="left" valign="top">OpenAI o4-mini</td><td align="left" valign="top">Reasoning</td><td align="left" valign="top">Text+image</td><td align="left" valign="top">Not disclosed</td><td align="left" valign="top">Enhanced visual reasoning; OCR not specified</td><td align="left" valign="top">Compact model with speed and accuracy; excels in technical tasks and visual reasoning</td><td align="left" valign="top">April 16, 2025</td></tr><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">GPT</td><td align="left" valign="top">Text, image, audio, and video</td><td align="left" valign="top">128,000 tokens</td><td align="left" valign="top">Strong multimodal; OCR not formally measured</td><td align="left" valign="top">Multimodal (voice, vision, and translation) with state-of-the-art benchmarks</td><td align="left" valign="top">May 13, 2024</td></tr><tr><td align="left" valign="top">GPT-5 (default)</td><td align="left" valign="top">GPT</td><td align="left" valign="top">Text, image, audio, and video</td><td align="left" valign="top">256,000 tokens</td><td align="left" valign="top">Enhanced visual perception; no explicit OCR metrics</td><td align="left" valign="top">Smart routing (fast vs deep mode), high performance in coding, writing, health, and perception</td><td align="left" valign="top">August 7, 2025</td></tr><tr><td align="left" valign="top">GPT-5 Thinking</td><td align="left" valign="top">GPT</td><td align="left" valign="top">Text, image, audio, and video</td><td align="left" valign="top">256,000 tokens</td><td align="left" valign="top">Tuned for deep reasoning; no explicit OCR mention</td><td align="left" valign="top">Tuned for deeper, slower reasoning; selected via router or user prompt</td><td align="left" valign="top">August 7, 2025</td></tr><tr><td align="left" valign="top">GPT-5 Pro</td><td align="left" valign="top">GPT</td><td align="left" valign="top">Text, image, audio, and video</td><td align="left" valign="top">256,000 tokens</td><td align="left" valign="top">Highest compute capability; no explicit OCR performance claim</td><td align="left" valign="top">High-compute, extended reasoning variant best suited for complex tasks</td><td align="left" valign="top">August 7, 2025</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>OCR: optical character recognition.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-6"><title>Outcomes and Evaluation Metrics</title><p>For Type D items, which were designed to assess the ability to recognize missing information, the expert reference answer was standardized as &#x201C;not recorded.&#x201D; Because models expressed absence in heterogeneous ways (eg, &#x201C;no record found&#x201D; and &#x201C;not noted in the chart&#x201D;), simple lexical comparisons could underestimate correct abstention. To address this, all responses explicitly indicating a lack of information were manually normalized to &#x201C;not recorded&#x201D; prior to scoring. This normalization procedure was applied consistently across all models and human raters.</p><p>Model outputs were compared against the gold-standard answers. Seven complementary metrics, all normalized to a continuous scale ranging from 0 to 1, were computed to capture both exactness and semantic adequacy:</p><list list-type="order"><list-item><p>Exact match (EM): the percentage of predictions that exactly match the reference answers, representing a strict, all-or-nothing measure. Values are 0 or 1; for free-text answers, a median of 0 is expected, with 1 indicating verbatim matches.</p></list-item><list-item><p>Token-level <italic>F<sub>1</sub></italic>-score: the harmonic mean of precision and recall calculated at the word (token) level, assessing overlap between predicted and reference text. Values range from 0 to 1, with values near 1 indicating close lexical overlap.</p></list-item><list-item><p>Bilingual evaluation understudy (BLEU): evaluates the quality of machine-generated text by measuring n-gram overlap between the prediction and a set of high-quality reference texts [<xref ref-type="bibr" rid="ref23">23</xref>]. Values range from 0 to 1 (higher is better); values above ~0.30 are considered understandable, and above ~0.50 are good and fluent [<xref ref-type="bibr" rid="ref24">24</xref>].</p></list-item><list-item><p>SacreBLEU: a standardized implementation of BLEU that ensures reproducible scores by handling tokenization and preprocessing consistently [<xref ref-type="bibr" rid="ref25">25</xref>]. It shares BLEU&#x2019;s 0&#x2010;1 scale and interpretation.</p></list-item><list-item><p>Recall-oriented understudy for gisting evaluation-longest common subsequence (ROUGE-L): measures the longest common subsequence between the generated and reference text, emphasizing recall [<xref ref-type="bibr" rid="ref26">26</xref>]. Values range from 0 to 1; like other overlap metrics, it yields near-zero values for free-text answers.</p></list-item><list-item><p>Bidirectional encoder representations from transformers (BERT) score: computes semantic similarity between candidate and reference text by aligning words using contextual embeddings from BERT [<xref ref-type="bibr" rid="ref27">27</xref>]. BERTScore values range from &#x2013;1 to 1, although in practice values fall within the 0&#x2010;1 range; roughly, &#x2265;0.90 is faithful, 0.66&#x2010;0.90 adequate, and &#x2264;0.65 meaning-divergent.</p></list-item><list-item><p>Sentence-BERT (SBERT): a modification of BERT using Siamese and triplet network architectures to generate semantically meaningful sentence embeddings, suitable for large-scale semantic comparison tasks [<xref ref-type="bibr" rid="ref28">28</xref>]. Cosine similarity ranges from &#x2212;1 to 1 but effectively falls within 0-1; commonly cited semantic-similarity thresholds fall in the 0.5&#x2010;0.7 range.</p></list-item></list></sec><sec id="s2-7"><title>Statistical Analysis</title><p>All automatic evaluation metrics were computed against the gold-standard answers on a per-question basis, with all items weighted equally (macro-averaging).</p><p>For each ChatGPT model, item-level scores were calculated across all 360 questions from the 15 charts. Distributions of EM, token-level <italic>F</italic><sub><italic>1</italic></sub>-score, BLEU, SacreBLEU, ROUGE-L, BERTScore, and SBERT were summarized per model using the median and IQR and visualized with boxplots. Model rankings and between-model comparisons were described descriptively.</p><p>To compare performance by task category, SBERT was prespecified as the primary metric. For each model and question type&#x2014;A (general), B (dental), C (interpretation), and D (absent information)&#x2014;item-level SBERT values were summarized using medians and IQRs and displayed as boxplots to facilitate within-type comparisons between models. This analysis included all 360 questions, with each item weighted equally.</p><p>For SBERT human-model comparisons, groups included Student 1, Student 2, Resident 1, Resident 2 (first-year residents), and the highest-performing GPT model. Analyses were conducted separately for question Types A-D. Because score distributions were nonnormal according to Shapiro-Wilk tests, overall group differences were assessed using the Kruskal-Wallis test, followed by Dunn post hoc tests with Holm-Bonferroni adjustment for pairwise contrasts of primary interest (each human group vs GPT-5 Thinking). All tests were 2-sided with a significance threshold of <italic>&#x03B1;</italic>=.05. Items with missing or empty reference answers were excluded listwise from the relevant comparisons. To convey practical significance alongside <italic>P</italic> values, effect sizes appropriate to the nonparametric design were computed: epsilon-squared (&#x03B5;&#x00B2;) for each Kruskal-Wallis omnibus test and Cliff delta (&#x03B4;) with 95% bootstrap CIs (10,000 resamples) for each prespecified pairwise contrast (each human group vs GPT-5 Thinking), interpreted with Cohen-equivalent benchmarks (&#x03B5;&#x00B2;=0.01/0.06/0.14; |&#x03B4;|=0.147/0.33/0.474 for the small/medium/large boundaries). SBERT embeddings for these comparisons were generated using sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2.</p><p>Interrater reliability of the gold-standard answers was assessed in 2 ways. First, an independent second senior specialist (JHK) in Conservative Dentistry reviewed all 360 gold-standard answers against the source charts and rated each as clinically acceptable. Second, because the senior faculty member and the 2 residents had answered the same items independently, each free-text answer was reduced to its canonical clinical value, and chance-corrected agreement was computed&#x2014;percent agreement, Krippendorff &#x03B1; (nominal), Fleiss &#x03BA;, and Gwet AC1. Gwet AC1 was prespecified as the primary coefficient because &#x03BA; can be unstable under near-ceiling agreement. Agreement was assessed overall and by question type for the clinical reference raters (gold standard and 2 residents) and, for transparency, for all 5 human raters. The canonical-value extraction was performed by an LLM used solely as an extractor (not as a judge of clinical correctness) and was validated against an independent dentist on a random 20% subset (50 items).</p><p>To assess whether the SBERT-based model ranking was not an artifact of embedding similarity, all 9 models were additionally scored for clinical correctness against the gold value (clinical content for Types A-C and appropriate abstention for Type D), and the resulting content-based clinical-accuracy ranking was compared with the SBERT ranking using Spearman rank correlation. As a safety analysis, the proportion of Type D items on which each model supplied a fabricated or unsupported value instead of abstaining was quantified. As a focused exploratory subanalysis, rather than a full multilingual evaluation, accuracy was compared between items whose expected answer was in Korean script and those whose expected answer was English or numeric, within a single task type. Because the charts contained both Korean and English within the same record, this comparison assessed the script of the expected answer rather than model performance across separate languages. All source code for model execution is available from the project&#x2019;s GitHub repository [<xref ref-type="bibr" rid="ref29">29</xref>].</p><p>To determine whether discrepancies flagged or missed by the automated metrics were clinically consequential, a prespecified expert-adjudicated subset analysis was performed for the 2 best-performing models. The adjudicated subset comprised all Type A-C responses scored as incorrect in the content-based clinical-accuracy rescoring (n=23), all Type D responses in which a model supplied a value instead of abstaining (n=6), and a random sample of 15 responses per model previously classified as content-correct with low SBERT similarity (&#x003C;0.60; n=30), the last stratum serving to verify that low semantic similarity did not conceal clinical errors (59 responses in total). Two faculty dentists at a university dental hospital, neither of whom had been involved in question authoring or gold-standard construction, independently rated each response in randomized order, blinded to model identity, automated scores, and subset stratum, on a 3-level scale: 0, no clinical consequence; 1, minor discrepancy unlikely to alter interpretation or management in a supervised educational setting; and 2, clinically consequential, defined as plausibly changing chart interpretation or a management decision or teaching an incorrect clinical fact. The interadjudicator agreement was quantified using Gwet AC1 and linearly weighted Cohen &#x03BA;, and disagreements were resolved by consensus.</p><p>All metric computations and statistical analyses were performed using Python (version 3.13.2; Python Software Foundation) on Windows 10 (Microsoft Corp).</p></sec><sec id="s2-8"><title>Exploratory Qualitative Feedback</title><p>As a secondary exploratory component, written qualitative feedback was collected from the 5 human participants (2 third-year dental students, 2 first-year residents, and the senior faculty dentist who established the gold standard) after they completed the chart-reading tasks. Participants responded in Korean to three open-ended questions concerning (1) the perceived advantages of AI-assisted dental chart interpretation for learning, (2) the perceived disadvantages or concerns, and (3) how the tool compared with traditional faculty-guided chart-reading instruction. Free-text responses were analyzed using a conventional content analysis approach. Two reviewers (MJJ and JI) independently read all responses, inductively coded recurrent ideas, and grouped them into themes for each question. The frequency of each theme was reported as the number of participants expressing it. Representative excerpts were translated from the original Korean and lightly edited for clarity. Because this component was exploratory and based on a small convenience sample, it was intended to contextualize, rather than to formally test, the quantitative benchmark findings. The full thematic summary and excerpts are provided in Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s2-9"><title>Ethical Considerations</title><p>The study protocol was approved by the Institutional Review Board (IRB) of the Seoul National University School of Dentistry (CRI26003). Because the study involved only fully deidentified historical dental chart images and no direct patient contact, informed consent was waived in accordance with the IRB guidelines. Voluntary informed consent was obtained from all participating dental students and clinicians prior to their involvement in the study. Data confidentiality and privacy were strictly maintained throughout the study.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>A total of 360 questions (15 charts&#x00D7;24 questions) were scored against the gold-standard answers. Model alignment, as measured by SBERT, generally exceeded that measured by string-based exactness metrics, indicating greater sensitivity to semantic adequacy than to exact matching. According to SBERT, the highest-performing model was GPT-5 Thinking, with a median score of 0.900 (IQR 0.525&#x2010;1.000), followed by GPT-5 Pro at 0.861 (IQR 0.501&#x2010;1.000) and OpenAI o3 at 0.831 (IQR 0.489&#x2010;1.000). Intermediate scores were observed for OpenAI o4-mini (0.695, IQR 0.444&#x2010;1.000) and GPT-4o (0.694, IQR 0.438&#x2010;1.000). Lower-performing models included GPT-4.5 (0.654, IQR 0.380&#x2010;0.874), GPT-5 (0.647, IQR 0.407&#x2010;0.940), GPT-4.1 (0.623, IQR 0.370&#x2010;0.861), and GPT-4.1 mini (0.529, IQR 0.311&#x2010;0.769). Within this framework, GPT-5 Thinking demonstrated the highest semantic alignment with reference answers, whereas GPT-4.1 mini exhibited the lowest. Full distributions across all 7 evaluation metrics are shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>, with per-item scores available in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Comparison of evaluation scores across multiple metrics for ChatGPT models and human responses. Colored boxplots represent item-level scores for each model and human group; higher values indicate closer agreement with reference answers. BERT: bidirectional encoder representations from transformers; BLEU: bilingual evaluation understudy; EM: exact match; ROUGE-L: recall-oriented understudy for gisting evaluation-longest common subsequence; SBERT: sentence bidirectional encoder representations from transformers.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e91809_fig03.png"/></fig><p>BLEU and SacreBLEU results reproduced the broad performance groupings identified by SBERT but revealed metric-specific reorderings among closely clustered models. ROUGE-L scores highlighted additional local inversions that differed from the SBERT rankings. These cross-metric differences emphasize that exactness-oriented and semantics-oriented metrics capture distinct types of model errors and should be interpreted jointly. In <xref ref-type="fig" rid="figure3">Figure 3</xref>, boxplots show that resident clinicians scored at or near the maximum for EM, <italic>F</italic><sub><italic>1</italic></sub>-score, BERTScore, and SBERT, whereas the LLM and student groups formed lower clusters, with near-zero medians for EM, SacreBLEU, and ROUGE-L.</p><p>Stratification by the 4 predefined question types (90 items per type) revealed task-dependent variability in model performance. <xref ref-type="fig" rid="figure4">Figure 4</xref> presents SBERT boxplots by model and question type. GPT-5 Thinking, GPT-5 Pro, and resident clinicians demonstrated higher and narrower score distributions for Types A and B, indicating consistent performance on explicit chart facts and tooth- or procedure-specific items. Type C (interpretation requiring multientry synthesis) showed more overlapping distributions across models, reflecting greater difficulty and variability. For Type D (absent information and abstention control), most groups achieved relatively high scores, indicating effective recognition of missing information. The SBERT-based relative rankings (highest to lowest) for each type were as follows.</p><list list-type="bullet"><list-item><p>Type A, general (explicit chart facts): GPT-5 Thinking, GPT-5 Pro, OpenAI o3, GPT-4o, OpenAI o4-mini, GPT-4.5, GPT-5, GPT-4.1, and GPT-4.1 mini</p></list-item><list-item><p>Type B, dental (tooth/procedure-specific): GPT-5 Thinking, GPT-5 Pro, GPT-5, OpenAI o3, GPT-4o, GPT-4.5, GPT-4.1, OpenAI o4-mini, and GPT-4.1 mini</p></list-item><list-item><p>Type C, interpretation (multientry synthesis): GPT-5, GPT-5 Thinking, GPT-4.5, GPT-4o, OpenAI o3, GPT-5 Pro, OpenAI o4-mini, GPT-4.1, and GPT-4.1 mini</p></list-item><list-item><p>Type D: absent information (abstention control): GPT-5 Pro, GPT-5 Thinking, OpenAI o3, GPT-4.1, GPT-4o, OpenAI o4-mini, GPT-4.5, GPT-5, and GPT-4.1 mini</p></list-item></list><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Boxplots of sentence bidirectional encoder representations from transformers (SBERT) scores by question type across ChatGPT models. Colored boxplots show item-level SBERT similarity scores for each model and human responses across question Types A-D (A, general; B, dental; C, interpretation; D, absent information); higher scores indicate closer semantic alignment with reference answers. Circles represent outliers. SBERT: sentence bidirectional encoder representations from transformers.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e91809_fig04.png"/></fig><p>For Types A, B, and D questions, the rank order generally mirrored the overall model analysis. Type A (fact retrieval) demonstrated high agreement with the reference answers; common factual items frequently achieved EM=1, reflecting reliable extraction of explicitly documented fields. In contrast, Type C (interpretation) produced compressed SBERT distributions, with medians ranging from 0.423 (IQR 0.278-0.692; GPT-4.1 mini) to 0.539 (IQR 0.321-0.738; GPT-5), a range of only 0.116 across models, and low EM rates, indicating that no model demonstrated a practically distinct advantage in multientry synthesis. Responses varied in evidence focus (eg, treatment plan, procedure note, or progress entry) and granularity (tooth-level or patient-level), resulting in heterogeneous rationales even when semantic similarity scores were closely clustered. For Type D, multiple models shared identical median scores; therefore, mean SBERT values were used to differentiate the rankings.</p><p>Illustrative examples are provided in <xref ref-type="table" rid="table3">Table 3</xref> and <xref ref-type="table" rid="table4">Table 4</xref>, showing a convergent Type B item with uniformly correct factual outputs and a divergent Type C item with heterogeneous rationales and reduced semantic similarity.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Example Type B (tooth-specific) question on the percussion finding of tooth number 36 at the initial visit, showing largely convergent factual answers and corresponding sentence bidirectional encoder representations from transformers (SBERT) similarity scores for human raters and ChatGPT models. SBERT similarity scores range from 0 to 1, with higher values indicating closer semantic agreement with the gold-standard answer. Question: What was the percussion finding on tooth number 36 at the initial visit?</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Model</td><td align="left" valign="top">Answer</td><td align="left" valign="top">SBERT similarity score (range 0-1)</td></tr></thead><tbody><tr><td align="left" valign="top">Gold Answer</td><td align="left" valign="top">per<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> (&#x2212;)</td><td align="left" valign="top">1.000</td></tr><tr><td align="left" valign="top">Resident 1</td><td align="left" valign="top">The tooth shows a negative percussion finding; per (&#x2212;)</td><td align="left" valign="top">0.652</td></tr><tr><td align="left" valign="top">Resident 2</td><td align="left" valign="top">At the initial visit, tooth number 36 showed a negative percussion response</td><td align="left" valign="top">0.187</td></tr><tr><td align="left" valign="top">Student 1</td><td align="left" valign="top">Negative</td><td align="left" valign="top">0.143</td></tr><tr><td align="left" valign="top">Student 2</td><td align="left" valign="top">(&#x2013;), no response to percussion</td><td align="left" valign="top">0.402</td></tr><tr><td align="left" valign="top">OpenAI o3</td><td align="left" valign="top">Negative percussion response</td><td align="left" valign="top">0.177</td></tr><tr><td align="left" valign="top">GPT-4.5</td><td align="left" valign="top">Percussion test positive (Pal<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>+, distal), bite test negative</td><td align="left" valign="top">0.035</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">At the initial visit, tooth number 36 showed percussion (&#x2013;), occlusal (&#x2013;), palatal (+), and bite (&#x2013;, bite stick)</td><td align="left" valign="top">0.283</td></tr><tr><td align="left" valign="top">GPT-4.1 mini</td><td align="left" valign="top">At the initial visit, tooth number 36 showed per (&#x2212;), bite (&#x2212;), pal (+), and distal sinus tract (+)</td><td align="left" valign="top">0.265</td></tr><tr><td align="left" valign="top">OpenAI o4-mini</td><td align="left" valign="top">At the initial examination, tooth number 36 showed a negative response to percussion; per (&#x2212;)</td><td align="left" valign="top">0.416</td></tr><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">At the initial visit, tooth number 36 showed a positive percussion response on palpation</td><td align="left" valign="top">0.294</td></tr><tr><td align="left" valign="top">GPT-5</td><td align="left" valign="top">At the initial visit, the percussion finding of tooth number 36 was pain on occlusion (&#x2212;), bite (&#x2212;), and pain on bite stick (&#x2212;)</td><td align="left" valign="top">0.187</td></tr><tr><td align="left" valign="top">GPT-5 Thinking</td><td align="left" valign="top">Per (&#x2013;)</td><td align="left" valign="top">0.569</td></tr><tr><td align="left" valign="top">GPT-5 Pro</td><td align="left" valign="top">Per (&#x2013;)</td><td align="left" valign="top">0.550</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>per: percussion.</p></fn><fn id="table3fn2"><p><sup>b</sup>pal: palpation.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Example Type C (interpretation) question on planned treatment for teeth numbers 37 and 47, illustrating divergent rationales and sentence bidirectional encoder representations from transformers (SBERT) similarity scores across human raters and ChatGPT models. SBERT similarity scores range from 0 to 1, with higher values indicating closer semantic agreement with the gold answer. Question: If treatment is performed on teeth numbers 37<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> and 47<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup>, what kind of treatment would be done?</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Model</td><td align="left" valign="top">Answer</td><td align="left" valign="top">SBERT similarity score (range 0-1)</td></tr></thead><tbody><tr><td align="left" valign="top">Gold answer</td><td align="left" valign="top">After performing a CLP<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup> by trimming the gum to increase the clinical crown length, caries treatment should be carried out, with RCT<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup> if necessary.</td><td align="left" valign="top">1.000</td></tr><tr><td align="left" valign="top">Resident 1</td><td align="left" valign="top">After performing CLP which trims the gum to increase the clinical crown length, caries treatment should be conducted with a possibility of RCT.</td><td align="left" valign="top">0.950</td></tr><tr><td align="left" valign="top">Resident 2</td><td align="left" valign="top">If necessary, caries treatment should be performed; however, CLP must be carried out beforehand, and there is a high likelihood of requiring RCT during the course of treatment.</td><td align="left" valign="top">0.942</td></tr><tr><td align="left" valign="top">Student 1</td><td align="left" valign="top">Distocervical caries can be removed and restored with resin.</td><td align="left" valign="top">0.242</td></tr><tr><td align="left" valign="top">Student 2</td><td align="left" valign="top">A gingivectomy or CLP may be required.</td><td align="left" valign="top">0.380</td></tr><tr><td align="left" valign="top">OpenAI o3</td><td align="left" valign="top">After a subgingival crown lengthening procedure, caries removal and conservative/restorative treatment, including root canal treatment if necessary, are required.</td><td align="left" valign="top">0.600</td></tr><tr><td align="left" valign="top">GPT-4.5</td><td align="left" valign="top">If treatment is performed on teeth numbers 37 and 47, there is a high likelihood that RCT will be required.</td><td align="left" valign="top">0.576</td></tr><tr><td align="left" valign="top">GPT-4.1</td><td align="left" valign="top">If treatment is performed on teeth numbers 37 and 47, regular monitoring of subgingival caries and, if necessary, conservative restorative treatment and root canal therapy will be considered.</td><td align="left" valign="top">0.206</td></tr><tr><td align="left" valign="top">GPT-4.1 mini</td><td align="left" valign="top">If treatment is performed on teeth numbers 37 and 47, a re-evaluation will be conducted after one year, with instructions on the use of interdental brushes and molar brushes, along with regular check-ups and additional treatment if necessary.</td><td align="left" valign="top">0.459</td></tr><tr><td align="left" valign="top">OpenAI o4-mini</td><td align="left" valign="top">If treatment is performed on teeth numbers 37 and 47, follow-up with bitewing X-rays at 6-month intervals is recommended. If symptoms or progression are observed, conservative restorative treatment or, if necessary, root canal therapy should be considered. It was also explained that careful case selection, including consideration of CLP, is required during treatment.</td><td align="left" valign="top">0.441</td></tr><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">It is explained that CLP is required beforehand, and the possibility of RCT is high.</td><td align="left" valign="top">0.805</td></tr><tr><td align="left" valign="top">GPT-5</td><td align="left" valign="top">High likelihood of RCT following CLP.</td><td align="left" valign="top">0.864</td></tr><tr><td align="left" valign="top">GPT-5 Thinking</td><td align="left" valign="top">After CLP, caries removal and restoration will be performed, with RCT if necessary.</td><td align="left" valign="top">0.888</td></tr><tr><td align="left" valign="top">GPT-5 Pro</td><td align="left" valign="top">Crown lengthening, followed by caries removal and restoration, with RCT if necessary.</td><td align="left" valign="top">0.548</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>37: left mandibular second molar.</p></fn><fn id="table4fn2"><p><sup>b</sup>47: right mandibular second molar.</p></fn><fn id="table4fn3"><p><sup>c</sup>CLP: crown-lengthening procedure.</p></fn><fn id="table4fn4"><p><sup>d</sup>RCT: root canal treatment.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Human-Model Comparison</title><p>The descriptive statistics (median and IQR) for each question type are summarized in <xref ref-type="table" rid="table5">Table 5</xref>, and the corresponding pairwise comparisons with GPT-5 Thinking, with effect sizes, are provided in Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. These comparisons are based on SBERT similarity to the gold standard, which reflects semantic adequacy rather than verified clinical correctness. Embedding similarity can rate clinically opposite findings as similar and clinically equivalent but differently worded answers as dissimilar. These contrasts, therefore, should be read as semantic proximity to the reference and interpreted alongside the content-based clinical accuracy analysis reported below. SBERT scores against the gold-standard answers were compared between GPT-5 Thinking and 4 human baselines&#x2014;Student 1 and Student 2 (preclinical-to-clinical transition students) and Resident 1 and Resident 2 (first-year residents)&#x2014;across question Types A-D (n=360), as shown in <xref ref-type="fig" rid="figure5">Figure 5</xref>. Because the distributions were nonnormal with unequal variances, Kruskal-Wallis tests indicated significant overall group differences for all 4 question types (all <italic>P</italic>&#x003C;.001; &#x03B5;&#x00B2;=0.052&#x2010;0.091). Dunn post hoc tests with Holm adjustment showed that Student 1 did not differ significantly from GPT-5 Thinking on Type A (<italic>P</italic>&#x2265;.99), Type B (<italic>P</italic>&#x2265;.99), Type C (<italic>P</italic>=.39), or Type D (<italic>P</italic>=.16). Student 2 differed only on Type D, scoring lower than GPT-5 Thinking (<italic>P</italic>=.002; &#x03B4;=&#x2212;0.27; 95% CI &#x2212;0.41 to &#x2212;0.13). Among resident clinician groups, Resident 1 scored higher than GPT-5 Thinking on Types A (<italic>P</italic>=.01; <italic>&#x03B4;</italic>=0.29; 95% CI 0.13-0.44) and C (<italic>P</italic>=.04; <italic>&#x03B4;</italic>=0.25; 95% CI 0.08-0.41) but did not differ significantly on Type B (<italic>P</italic>=.07) and scored lower on Type D (<italic>P</italic>&#x003C;.001; &#x03B4;=&#x2212;0.35; 95% CI &#x2212;0.49 to &#x2013;0.21). Resident 2 scored higher than GPT-5 Thinking on Types A (<italic>P</italic>=.03; <italic>&#x03B4;</italic>=0.25; 95% CI 0.09-0.41), B (<italic>P</italic>=.02; <italic>&#x03B4;</italic>=0.28; 95% CI 0.11-0.44), and C (<italic>P</italic>=.04; <italic>&#x03B4;</italic>=0.26; 95% CI 0.09-0.42), with no significant difference on Type D (<italic>P</italic>=.23). Overall, effect sizes were negligible to small except for the Resident 1 Type D contrast, which was medium in magnitude. These results indicate that GPT-5 Thinking was broadly comparable to the students across most tasks and remained below resident-level performance for several Types A-C contrasts. Type D warrants particular caution. Because all groups shared identical ceiling medians and IQRs of 1.000, the 2 significant contrasts arose only from a few lower-scoring tail items, reflecting distributional subtleties rather than a real difference in recognizing absent information.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Comparison of sentence bidirectional encoder representations from transformers (SBERT) similarity to the gold-standard answers between GPT-5 Thinking and human raters, by question type.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question type</td><td align="left" valign="bottom">GPT-5 Thinking</td><td align="left" valign="bottom">Resident 1</td><td align="left" valign="bottom">Resident 2</td><td align="left" valign="bottom">Student 1</td><td align="left" valign="bottom">Student 2</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">SBERT similarity to gold standard, median (IQR)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type A (general)</td><td align="left" valign="top">0.978 (0.870&#x2010;1.000)</td><td align="left" valign="top">1.000 (1.000&#x2010;1.000)</td><td align="left" valign="top">1.000 (0.978&#x2010;1.000)</td><td align="left" valign="top">0.994 (0.863&#x2010;1.000)</td><td align="left" valign="top">0.904 (0.744&#x2010;1.000)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type B (dental)</td><td align="left" valign="top">0.813 (0.525&#x2010;0.969)</td><td align="left" valign="top">0.968 (0.616&#x2010;1.000)</td><td align="left" valign="top">0.991 (0.649&#x2010;1.000)</td><td align="left" valign="top">0.642 (0.336&#x2010;1.000)</td><td align="left" valign="top">0.744 (0.404&#x2010;1.000)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type C (interpretation)</td><td align="left" valign="top">0.556 (0.371&#x2010;0.650)</td><td align="left" valign="top">0.641 (0.478&#x2010;0.861)</td><td align="left" valign="top">0.688 (0.413&#x2010;0.820)</td><td align="left" valign="top">0.478 (0.221&#x2010;0.607)</td><td align="left" valign="top">0.406 (0.271&#x2010;0.580)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type D (absent information)</td><td align="left" valign="top">1.000 (1.000&#x2010;1.000)</td><td align="left" valign="top">1.000 (1.000&#x2010;1.000)</td><td align="left" valign="top">1.000 (1.000&#x2010;1.000)</td><td align="left" valign="top">1.000 (1.000&#x2010;1.000)</td><td align="left" valign="top">1.000 (1.000&#x2010;1.000)</td></tr><tr><td align="left" valign="top" colspan="6">Dunn post hoc comparison versus GPT-5 Thinking (<italic>P</italic> value)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type A (general)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">.01</td><td align="left" valign="top">.03</td><td align="left" valign="top">&#x003E;.99</td><td align="left" valign="top">.37</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type B (dental)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">.07</td><td align="left" valign="top">.02</td><td align="left" valign="top">&#x003E;.99</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type C (interpretation)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">.04</td><td align="left" valign="top">.04</td><td align="left" valign="top">.39</td><td align="left" valign="top">.13</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Type D (absent information)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">.23</td><td align="left" valign="top">.16</td><td align="left" valign="top">.002</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Boxplots of sentence bidirectional encoder representations from transformers (SBERT) scores by question type for GPT-5 Thinking and human responses. Colored boxplots show item-level SBERT scores for GPT-5 Thinking, Resident 1 and Resident 2, and Student 1 and Student 2 (third-year dental students). SBERT: sentence bidirectional encoder representations from transformers.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e91809_fig05.png"/></fig></sec><sec id="s3-3"><title>Reliability of the Gold Standard</title><p>The independent senior specialist judged all 360 gold-standard answers to be clinically acceptable (100% endorsement). Among the clinical reference raters, agreement on the extracted clinical value was very high (Gwet AC1=0.99 overall and &#x2265;0.98 for every task type; Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>), indicating that the gold standard was reproducible across independent, qualified clinicians. The canonical-value extraction agreed with an independent dentist on 97.2% of the validation subset. When the 2 third-year students were included, agreement was lower (Gwet AC1=0.83; Table S4 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>), reflecting their developing chart-reading competence rather than the unreliability of the reference. SBERT cosine and BERTScore <italic>F</italic><sub><italic>1</italic></sub>-score showed the same ordering at lower magnitudes (clinical reference raters &#x2248; 0.85 and 0.87) and are reported only as descriptive checks.</p></sec><sec id="s3-4"><title>Robustness of the Model Ranking and Safety</title><p>To confirm that the ranking reflected clinical content rather than embedding similarity alone, all 9 models were independently rescored for clinical correctness against the gold-standard value. This content-based clinical-accuracy ranking was closely aligned with the SBERT ranking (Spearman &#x03C1;=0.90; <italic>P</italic>=.001; Table S5 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>), with GPT-5 Pro, GPT-5 Thinking, and OpenAI o3 forming the leading group (clinical accuracy 96.4%, 95.6%, and 91.4%, respectively) and GPT-4.1 mini the lowest (68.9%). This convergence indicates that the model ordering was not an artifact of semantic-similarity scoring. On the absent-information (Type D) task, the proportion of fabricated answers in Table S6 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>, supplying a value the chart did not contain instead of abstaining, rose as model capability fell, from 2.2% (GPT-5 Pro) and 4.4% (GPT-5 Thinking) to 28.9% (GPT-5) and 46.7% (GPT-4.1 mini), whereas all 3 clinical raters abstained appropriately on every such item. This focused exploratory comparison, limited to the tooth- and procedure-specific task, showed no significant difference between Korean-script and English or numeric answers (87.8% vs 90.4%; <italic>P</italic>=.22), indicating that the mixed language input did not by itself drive performance differences within this task. Given its narrow scope, this result is descriptive and does not constitute a comprehensive multilingual evaluation.</p></sec><sec id="s3-5"><title>Expert Adjudication of Clinically Consequential Discrepancies</title><p>The exact agreement between the 2 adjudicators was 91.5% (n=59; Gwet AC1=0.879; linearly weighted &#x03BA;=0.89), and all 5 disagreements were between adjacent categories and were resolved by consensus. Of the 59 adjudicated responses, 10 were rated as clinically consequential: 4 of 23 were content-incorrect responses, and all 6 were Type D fabrications. Within the adjudicated subset, clinically consequential ratings occurred only among responses previously classified as content-incorrect or fabricated. The 10 identified cases represented 1.4% of the 720 outputs generated by the 2 models; however, because adjudication was subset-based, this percentage should not be interpreted as a full-sample incidence estimate. None of the 30 sampled low-SBERT responses previously classified as content-correct was rated as clinically consequential, indicating that low SBERT similarity in this sampled stratum reflected wording differences rather than clinical error, as shown in Table S7 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s3-6"><title>Exploratory Qualitative Feedback</title><p>Content analysis of the open-ended responses from the 5 participants yielded a small set of recurrent themes for each question, summarized with representative excerpts in Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. For perceived advantages, time efficiency was noted by all 5 participants (n=5), followed by accessibility, immediate feedback, repeated learning, and support for self-directed learning (each n=3). For perceived disadvantages, the possibility of AI errors or inaccuracies was raised by all participants (n=5), with the lack of clinical context, overreliance associated with reduced critical thinking, and privacy or data-leakage risks each noted by 2 participants (n=2). In comparing the tool with traditional faculty-guided instruction, most participants (n=4) viewed AI as a supplementary tool to be used alongside faculty rather than as a replacement, and 3 participants (n=3) considered it particularly useful during the early learning stages for terminology and basic chart reading. Notably, 4 participants (n=4) emphasized that faculty expertise remains necessary when contextualized clinical reasoning or treatment decisions are required, indicating that participants regarded AI as most appropriate for foundational, evidence-locatable tasks while reserving complex clinical judgment for faculty guidance.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>In this image-based benchmark of dental chart interpretation, the main findings were 3-fold. First, reasoning-optimized models led across tasks: GPT-5 Thinking achieved the highest overall semantic alignment with the gold standard (SBERT median 0.900), followed by GPT-5 Pro and OpenAI o3, while the smallest models performed worst, and a content-based clinical-accuracy re-scoring reproduced this ranking (Spearman &#x03C1;=0.90; <italic>P</italic>=.001). Second, in the human comparison, GPT-5 Thinking was statistically indistinguishable from dental students on most question-type contrasts but remained below first-year residents on several factual, dental-specific, and interpretive (Types A-C) contrasts, with multientry synthesis (Type C) being the most challenging task across groups. Third, abstention on absent-information (Type D) items was generally reliable, although fabrication rose sharply as model capability fell (from 2.2% for GPT-5 Pro to 46.7% for GPT-4.1 mini), whereas all clinical raters abstained appropriately on every such item. Together, these results position the strongest multimodal models as potential supervised aids for chart-reading practice rather than as substitutes for clinical expertise.</p><p>Each question paired a dental chart image with a text prompt, requiring models to parse chart layouts, interpret symbolic markings, and align visual cues with textual questions, similar to emerging evaluations of medical multimodal LLMs in radiology and nuclear medicine, where vision&#x2013;language integration and domain grounding are critical determinants of performance [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref33">33</xref>]. Consistent with reviews noting that LLM outcomes are strongly task- and modality-dependent and that safety and hallucination control are key concerns in clinical settings [<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref38">38</xref>], the heterogeneous patterns across models and task types are compatible with differences in reasoning architecture, safety alignment, and multimodal training described in GPT-5 system documentation [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>].</p><p>Across Type A items (explicit chart-based facts), GPT-5 Thinking, GPT-5 Pro, and OpenAI o3 formed the leading group. For Type B items (tooth- and procedure-specific reasoning), GPT-5 Thinking and GPT-5 Pro again ranked highest, followed by GPT-5. Public documentation describes GPT-5 Thinking as a reasoning-focused variant trained with reinforcement-learning methods and &#x201C;safe completions&#x201D; designed to reduce unsafe or overconfident outputs and GPT-5 Pro as a configuration with additional test-time compute [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. These design characteristics align with the relatively stable factual performance observed, though other factors, such as visual processing pipelines and context handling, may also contribute [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref33">33</xref>]. For Type C items, which required synthesis across multiple chart regions and textual entries, GPT-5 outperformed GPT-5 Thinking and GPT-5 Pro. One possible interpretation is that general-purpose tuning and strong long-context capabilities facilitated flexible integration of disparate findings [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>], whereas the more cautious, stepwise reasoning of GPT-5 Thinking and GPT-5 Pro may have limited the breadth of synthesized interpretations. However, this post hoc explanation cannot be definitively separated from potential influences of context-window use, input batching, or image-handling behavior. Type D items assessed abstention behavior in the presence of missing information. GPT-5 Pro demonstrated the most reliable abstention performance, followed by GPT-5 Thinking and OpenAI o3, whereas GPT-5 performed comparatively poorly. Reasoning-focused models such as GPT-5 Thinking are reported to prioritize safety and adherence to instructions through reinforcement learning and safe completions [<xref ref-type="bibr" rid="ref39">39</xref>]. The observed pattern is consistent with more conservative responses and aligns with prior work on abstention and hallucination control in safety-tuned models [<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]; however, this does not constitute direct evidence that specific training procedures caused the observed behavior.</p><p>Aggregated across task types, the overall ranking was GPT-5 Thinking, GPT-5 Pro, OpenAI o3, OpenAI o4-mini, GPT-4o, GPT-4.5, GPT-5, GPT-4.1, and GPT-4.1 mini. The leading positions of GPT-5 Thinking and GPT-5 Pro in this multimodal dental setting suggest that, across a mixture of factual, procedural, integrative, and abstention-focused tasks, cautious structured reasoning and calibrated uncertainty may contribute more to overall utility than maximally flexible generation. The strong performance of OpenAI o3 is consistent with its role as a reasoning-focused predecessor [<xref ref-type="bibr" rid="ref39">39</xref>]. The relatively high placement of OpenAI o4-mini indicates that some questions&#x2014;particularly those requiring explicit factual recognition or simple abstention&#x2014;can be addressed effectively by more parameter-efficient models, consistent with findings in medical multimodal imaging [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref33">33</xref>]. Between-model differences may, however, reflect variation in context-window size, segmented uploads of large charts, differences in OCR and image-processing pipelines, and heterogeneous handling of Korean-language inputs, cautioning against strong claims about intrinsic capability. These findings tentatively support a model-by-task strategy, to be validated prospectively. Within the demanding Type B task, Korean-script and English or numeric answers did not differ (87.8% vs 90.4%; <italic>P</italic>=.22), indicating that code-mixed input did not by itself drive the results.</p><p>Against human raters, GPT-5 Thinking was broadly comparable to students and below first-year residents on several extraction and interpretive contrasts. Most student-model comparisons were not statistically significant, although Student 2 scored lower on the absent-information task. Practically, for basic extraction and description of information, GPT-5 Thinking matched students transitioning to clinical training, while remaining below residents on several Type A-C contrasts, consistent with prior education evaluations in which ChatGPT-level systems often reach, but rarely surpass, student performance on written examinations and vignette-style assessments [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref46">46</xref>]. This pattern was clearest on the more &#x201C;dentally demanding&#x201D; tasks&#x2014;Type B (tooth- and procedure-specific reasoning) and Type C (multientry synthesis)&#x2014;where resident clinicians generally had higher scores than GPT-5 Thinking and students, suggesting differences in prioritization and contextual interpretation. On Type D, all groups clustered near the high end of the score range, frequently recognizing missing information and avoiding overinterpretation, and reached identical ceiling medians. Where pairwise contrasts nonetheless reached significance, the differences stemmed from a few tail items rather than from a systematic gap in abstention and are best read as distributional rather than clinically meaningful. Although GPT-5-family models are described as incorporating approaches intended to reduce hallucination and promote conservative responses under uncertainty [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref44">44</xref>], this finding provides reassurance that the model often adopts a safe abstention strategy when key information is missing, though it does not imply parity with clinicians. Quantitatively, the proportion of fabricated answers on absent-information items ranged from 2.2% (GPT-5 Pro) and 4.4% (GPT-5 Thinking) to 28.9% (GPT-5) and 46.7% (GPT-4.1 mini), whereas all 3 clinical raters abstained appropriately on every item, underscoring that the cheapest models are unsafe for unsupervised use. Overall, GPT-5 Thinking is most appropriately framed as a supervised support tool or virtual peer learner, not a substitute for clinician judgment [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref37">37</xref>].</p><p>The 7 metrics captured different aspects of model behavior. Exactness-oriented metrics (EM, ROUGE-L, and SacreBLEU) were largely uninformative, reaching nonzero medians essentially only for residents because they reward verbatim reproduction rarely expected in open-ended explanations. Lexical-overlap metrics (<italic>F</italic><sub><italic>1</italic></sub>-score, BLEU) gave a clearer but still surface-level hierarchy (residents &#x003E;GPT-5 Thinking &#x003E;OpenAI o3 &#x2248; GPT-5 Pro &#x003E;remaining models and students). Semantic metrics were most informative: on SBERT, resident clinicians scored 1.0, GPT-5 Thinking 0.900, GPT-5 Pro 0.861, and OpenAI o3 0.831, with students around 0.77&#x2010;0.78 and other models lower. Because these embedding metrics track sentence-level meaning rather than wording, GPT-5 Thinking, GPT-5 Pro, and OpenAI o3 constitute a &#x201C;reasoning-optimized&#x201D; cluster that is semantically closer to resident clinician answers than both students and earlier models. Within the GPT-5 family, the gap between GPT-5 and GPT-5 Thinking further implies that architectural and alignment choices aimed at deeper reasoning translate into measurable gains in clinical content quality, beyond what can be achieved by scale alone [<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref44">44</xref>]. Median scores for GPT-5 Thinking fell between students and residents; wide IQRs reflect varying difficulty of the clinical questions, the probabilistic nature of LLM responses, and the individual differences in clinical experience among the students. Importantly, embedding similarity reflects semantic adequacy rather than clinical correctness: it can rate clinically opposite findings as highly similar (eg, per [+] vs per [&#x2212;]) and clinically equivalent statements as dissimilar when their wording differs. SBERT is therefore best interpreted as a scalable proxy for semantic proximity to the reference, not as a direct measure of clinical correctness, and all SBERT-based human-model contrasts should be interpreted accordingly. To guard against this limitation, the SBERT ranking was corroborated by an independent content-based clinical-accuracy rescoring of all 9 models, which reproduced the same ordering (Spearman &#x03C1;=0.90; <italic>P</italic>=.001); this content-based analysis, rather than embedding similarity alone, anchors the interpretation of relative model performance. Results should still be read cautiously given the modest, single-institution sample.</p><p>The expert-adjudicated subset analysis reinforces this reading of the automated metrics. Within the adjudicated subset, clinically consequential ratings occurred only among responses previously flagged as content-incorrect or fabricated, and none occurred in the sampled low-similarity responses previously classified as content-correct. The 10 identified cases represent 1.4% of the 720 outputs generated by the 2 models; however, because adjudication was subset-based, this percentage should not be interpreted as a full-sample incidence estimate. Because adjudication was limited to the 2 best-performing models and used a rubric-based consequence scale rather than observed educational outcomes, these findings characterize the adjudicated subset within this benchmark rather than clinical performance.</p></sec><sec id="s4-2"><title>Implications for Dental Education</title><p>From an educational standpoint, the findings suggest that GPT-5 Thinking could potentially serve as a supervised reference and secondary-reader tool within chart-reading training. It is particularly suited for exercises emphasizing evidence identification, omission detection, and uncertainty recognition, skills central to safe documentation and clinical reasoning. Given the safety risks associated with unsupervised LLM errors in direct patient care, focusing on chart interpretation rather than definitive diagnosis better reflects the potential utility of LLMs as safe cognitive training tools in risk-free educational settings. Because the model matched novice learners on most contrasts but stayed below residents on several Type A-C tasks (notably Type B and C), human supervision remains essential. Rather than replacing instructor guidance, multimodal LLMs can facilitate guided education, allowing students to query chart content, test hypotheses, and receive structured feedback under supervision. This perspective aligns with automation-bias literature emphasizing verification and guardrails [<xref ref-type="bibr" rid="ref47">47</xref>] and with the findings by Hattie and Timperley [<xref ref-type="bibr" rid="ref48">48</xref>] that high-quality feedback is an important influence on achievement, suggesting that multimodal LLMs may be most beneficial when integrated into feedback-rich, supervised learning activities.</p><p>Practical integration into dental curricula can be achieved through digital exercises using deidentified charts, case-based seminars, and self-directed study tasks. These activities should emphasize locating relevant entries, reconstructing treatment timelines, and verifying documented procedures, thereby helping students develop both chart-reading literacy and AI literacy. Advances in multimodal and interactive interfaces further align training with clinical practice, allowing students to upload chart images, use voice input during simulations, and receive immediate feedback [<xref ref-type="bibr" rid="ref49">49</xref>].</p><p>Safe and responsible adoption requires institutional policies that define permissible uses, mandate verification against original charts, and record model details for assignments. The use of deidentified data and audit logs should be standard to ensure transparency and accountability. Within these guardrails, multimodal LLMs can function as collaborative aids in guided education, promoting deliberate practice and structured feedback while maintaining professional oversight. This approach aligns with Masters&#x2019; recommendations for ethical AI use in health professions education, which emphasize governance, transparency, and accountability frameworks [<xref ref-type="bibr" rid="ref50">50</xref>].</p><p>An exploratory qualitative feedback analysis obtained from study participants and summarized in Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> provided additional insight into the perceived educational usefulness of AI-assisted dental chart interpretation. Participants consistently identified time efficiency, accessibility, immediate feedback, and support for self-directed learning as key advantages of the tool. These findings are consistent with emerging evidence suggesting that generative AI can function as an on-demand educational resource that enhances learner autonomy and supports personalized learning in health professions education [<xref ref-type="bibr" rid="ref51">51</xref>]. Prior studies have also suggested that LLMs may serve as valuable educational aids by providing rapid access to information, supporting formative feedback, and facilitating independent learning outside traditional instructional settings [<xref ref-type="bibr" rid="ref8">8</xref>]. Furthermore, AI tools may function as cognitive scaffolds that help learners navigate complex clinical information while promoting active engagement with educational content and supporting the development of foundational competencies [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. Notably, participants perceived the tool as useful in early clinical training, when learners are first exposed to unfamiliar terminology, chart structures, and clinical documentation. This finding suggests that multimodal AI may facilitate the preclinical-to-clinical transition by providing timely support and opportunities for deliberate practice.</p><p>Participants also expressed concerns about inaccuracies, overreliance, privacy, and the inability of AI systems to fully capture contextual reasoning. Importantly, both students and residents regarded AI as a supplementary educational tool rather than a replacement for faculty-guided instruction, particularly when complex clinical reasoning and judgment were required. These perceptions align with recent discussions in medical education emphasizing that generative AI should augment rather than replace human expertise and that learners must be trained to critically evaluate AI-generated outputs [<xref ref-type="bibr" rid="ref53">53</xref>]. Concerns regarding cognitive dependency and reduced independent reasoning have likewise been highlighted as important challenges associated with the educational use of generative AI [<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>]. Together, these findings support multimodal AI as an adjunctive learning technology that enhances chart-reading literacy under supervision while preserving the central educational role of faculty.</p></sec><sec id="s4-3"><title>Comparison With Prior Work</title><p>This study makes 2 novel contributions. First, it represents the first systematic evaluation of multimodal LLMs for dental chart image interpretation. Second, it implements a multiaxis performance assessment across 4 distinct task types, including general fact retrieval (Type A), tooth- or procedure-level retrieval (Type B), interpretive and inference reasoning (Type C), and absent-information and abstention control (Type D).</p><p>Prior studies have primarily examined LLMs&#x2019; ability to extract information from text-based inputs, such as discharge summaries or narrative clinical notes. While these tasks assess factual comprehension, they omit the spatial information inherent in clinical records, particularly in dental charts that include odontograms, tooth-number grids, and procedure maps. Linearizing such records into text disrupts the contextual relationships between spatial elements and their annotations. Evaluating image-based chart inputs, therefore, probes multimodal reasoning under more realistic conditions, reflecting how clinicians interpret dental charts by integrating both spatial and textual cues. Although multimodal capability has been documented in several medical imaging domains, such as radiography and pathology slide analysis [<xref ref-type="bibr" rid="ref30">30</xref>], comparable evaluations using dental chart images have not previously been reported, motivating the present image-based benchmarking.</p><p>With respect to the second contribution, most prior evaluations have relied on multiple-choice or short-answer formats, such as the United States Medical Licensing Examination (USMLE), reporting aggregate accuracy without decomposing performance by task type [<xref ref-type="bibr" rid="ref8">8</xref>]. The present findings align with previous reports indicating that LLMs can reach or exceed passing thresholds on structured, well-specified examinations but remain less reliable on open-ended reasoning and fact-checking tasks [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Reviews have similarly emphasized hallucination and inconsistent abstention as persistent limitations, highlighting the need for human verification and explicit guardrails in both educational and clinical contexts [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref55">55</xref>]. In dental education, a mixed methods study of ChatGPT in undergraduate training reported benefits for rapid clarification, idea generation, and formative feedback, while also noting factual errors and over-reliance, prompting recommendations for supervised, source-verified use [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>Language effects further contextualize the present results. Earlier work suggested higher LLM performance in English compared with other languages; however, recent multimodal models (eg, GPT-4o) have narrowed this gap, demonstrating improved correctness and responsiveness in Japanese across general and medical imaging tasks, including radiology [<xref ref-type="bibr" rid="ref56">56</xref>]. In dentistry, a peer-reviewed evaluation of GPT-4o and Gemini Advanced on the Korean National Dental Licensing Examination documented strong accuracy and consistency on well-specified, evidence-locatable items [<xref ref-type="bibr" rid="ref57">57</xref>]. Emerging assessments of GPT-5 models in Korean similarly report robust performance on structured, multimodal tasks, findings broadly consonant with the present study&#x2019;s mixed Korean and English chart data. Collectively, these studies suggest that recent multimodal LLM families are reducing prior language penalties, particularly when queries are context-bounded and answers can be anchored to identifiable evidence. The present language comparison, however, was a focused exploratory subanalysis within a single task rather than a controlled multilingual evaluation, so it can only contextualize these findings and cannot establish language equivalence.</p></sec><sec id="s4-4"><title>Practical Guidance for Use</title><p>The highest-performing models in the study, such as GPT-5 Thinking and GPT-5 Pro, achieved the greatest accuracy but required longer response times. In time-sensitive educational contexts, such as self-directed chart-reading practice or preparation for case-based seminars, faster models like GPT-4o or OpenAI o4-mini may be more practical, as modest reductions in accuracy are offset by faster responses [<xref ref-type="bibr" rid="ref58">58</xref>]. Two key implications emerge for educational use. First, model selection should be guided by task demands: faster models are suitable for routine information retrieval during practice, whereas reasoning-tuned models are preferable when accuracy and abstention are critical. Second, all outputs should be verified against the original record before clinical use, given the potential for errors when documentation is incomplete or absent [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Because the models were accessed through the consumer interface, exact compute cost and latency were not measured. Any cost-related guidance is therefore provisional and based on accuracy alone. On that basis, GPT-5 Thinking reached clinical accuracy close to the larger GPT-5 Pro, suggesting a more cost-effective option for educational deployment, whereas the smallest models, though inexpensive, abstained poorly and are not advisable for unsupervised use. These observations should be confirmed by a formal API-based benchmark of cost and latency alongside accuracy, which remains an important next step. For educators and learners considering practical use, the main conclusions can be summarized at a glance: across factual, dental-specific, interpretive, and abstention tasks, the strongest reasoning-optimized models (GPT-5 Thinking, GPT-5 Pro, and OpenAI o3) matched novice-student retrieval and abstained reliably, yet remained below resident-level interpretive reasoning, and fabrication of absent information rose steeply as model capability declined (from 2.2% to 46.7%). The practical guidance is therefore consistent across all 4 task types&#x2014;multimodal LLMs are best used as supervised, verification-focused aids whose outputs are checked against the original record, rather than as autonomous interpreters. Although demonstrated here for dental records, this guidance parallels emerging questions about record-reading and documentation literacy in medical and nursing practice; whether the same supervised, verify-before-use framework transfers to other specialties or to adjacent health-profession settings remains an open question requiring external validation.</p><p>Although this study focused on educational benchmarking, one possible area for future investigation is whether multimodal LLMs could support clinical information retrieval from lengthy, mixed-language dental records. For example, identifying the date of a composite resin restoration on the upper right first premolar across multiple years of visits typically requires extensive manual review. A single structured query to an LLM can quickly surface the relevant note and date for confirmation, reducing search time and cognitive load. However, as this clinical-retrieval scenario was outside the scope of the present benchmark, it is best regarded as a direction for future prospective, safety-focused studies.</p></sec><sec id="s4-5"><title>Limitations</title><p>The dataset was drawn from a single department (Conservative Dentistry) from a single institution (SNUDH) and included only 15 charts. This study should therefore be regarded as an internal, single-center benchmark. Because multimodal chart interpretation depends heavily on local layout, documentation conventions, abbreviations, and specialty-specific workflows, which vary across institutions and electronic dental record systems, the single-center design carries an inherent risk of institutional bias, and the findings may not be directly generalizable to other specialties, institutions, record systems, or linguistic settings without external validation. Records contained both Korean and English entries, and results may differ in other language contexts.</p><p>The analysis reflected model versions selectable in the consumer interface under a Pro-tier subscription in early August 2025; model availability is both interface- and subscription-dependent and may differ at other times, under different tiers, or via API access. In addition, the models were accessed through the consumer interface, so exact token usage, latency, and compute cost were not measured.</p><p>Human comparisons were limited to 2 students and 2 residents, as their primary role was to establish a reference baseline for the AI models rather than to serve as a large-scale population sample. Nevertheless, this reduces the statistical precision of group contrasts and warrants confirmation in studies with larger human cohorts.</p><p>A further consideration concerns the independence of observations. Because the 24 questions for each patient were derived from the same chart, items from the same chart may share case-level characteristics such as complexity, documentation style, and image quality, so their scores are not fully independent. The nonparametric tests applied here treat items as independent and do not account for this within-chart clustering. The effective sample size is therefore smaller than the nominal number of items, and the item-level comparisons should be regarded as exploratory. Confirmatory analyses using clustered or mixed effects models with the chart as a grouping factor would help verify the robustness of these findings.</p><p>Reliability and scoring relied on semantic and content-based proxies rather than a formal clinical-harm rubric; although SBERT was triangulated with content-level agreement and canonical-value extraction was validated against an independent dentist on a 20% subset (97.2% agreement), residual error cannot be excluded. The expert-adjudicated subset analysis addressed clinically consequential discrepancies within this benchmark (Table S7 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>), but prospective grading of clinical harm under real educational use, by a larger expert panel, remains important future work.</p><p>In addition, the 360 questions were authored by a single clinical researcher (AYC) and were not independently reviewed for wording before evaluation, which may introduce authoring bias. This is most relevant for the interpretive Type C items, where multiple clinically acceptable framings and differing evidence-prioritization strategies can exist, potentially shaping item difficulty, ambiguity, and, in principle, model ranking. Several design features mitigate this concern&#x2014;the identical question set was applied to all models, the gold-standard answers were independently judged clinically acceptable with high interrater agreement (Gwet AC1=0.99), and the SBERT-based ranking was reproduced by a content-based clinical-accuracy rescoring (Spearman &#x03C1;=0.90)&#x2014;but confirmation with independently reviewed or multiauthor question sets remains warranted.</p><p>Several methodological constraints also warrant caution. First, because evaluation was end-to-end, OCR accuracy was not independently measured, and the chart resolution (717&#x00D7;742 pixels), the maximum obtainable resolution from the EMR interface, may have constrained visual parsing; this nonetheless reflects the native, constrained outputs that students and clinicians actually use, making robustness to such inputs relevant to real-world utility. Second, evaluations used the consumer graphical user interface (GUI) rather than the API, and the study was not preregistered, as it was an exploratory benchmark. This reflects a deliberate trade-off between ecological validity and computational reproducibility [<xref ref-type="bibr" rid="ref59">59</xref>]. The GUI captures how students and clinicians actually interact with these models, but it introduces GUI-specific uncontrolled variables such as proprietary system prompts, memory-state handling, rate limiting, and dynamic model routing that are not under experimental control. These factors are further amplified for the larger charts, whose images exceeded the 10-image message limit and were therefore uploaded in consecutive batches within a single conversation thread. Consequently, the present results should be interpreted as real-world interface benchmarking rather than fully controlled model benchmarking. Because the behavior of the same commercial LLM service has been shown to shift substantially over short periods as providers update it without public notice [<xref ref-type="bibr" rid="ref60">60</xref>], the observed model rankings may not be fully stable under future platform updates or under API-based, version-pinned evaluation. Finally, as a comparative study with students, this study cannot demonstrate whether using these models improves learning outcomes. Future work should use larger, multi-institutional, multispecialty, and ideally multinational datasets, complement this ecological benchmark with API-based, version-pinned, preregistered evaluations, and incorporate formal clinical grading rubrics to assess educational impact, safety, response time, and efficiency under real clinical conditions.</p></sec><sec id="s4-6"><title>Conclusions</title><p>This study compared multimodal ChatGPT models with dental students and first-year residents in reading and interpreting dental chart images. GPT-5 Thinking demonstrated the highest overall SBERT performance and was broadly comparable to dental students, with no significant difference from Student 1 across Types A-D and only one significant contrast with Student 2 on Type D. Compared with residents, however, it remained lower on several Type A-C contrasts, particularly Type C and one Type B comparison; Type D findings should be interpreted cautiously because all groups had ceiling medians. These findings indicate that current multimodal models can approach student-level comprehension in structured chart reading but remain below resident-level reasoning in several resident comparisons. In educational settings, such models may serve as supervised support tools for practicing chart interpretation and verification. These findings derive from an internal, single-center benchmark, and external validation across multiple institutions, dental specialties, record systems, and language settings is required before broader claims can be made. As multimodal capabilities continue to advance, carefully integrated AI systems may have the potential to support early clinical training through guided education, offering scalable, feedback-driven support to strengthen chart-reading literacy.</p></sec></sec></body><back><ack><p>The authors thank the Dental Research Institute for its support and the students and residents who participated in and assisted with this study. The authors also thank Seoul National University Dental Hospital for providing access to the clinical records used in this research. We used AI tools (ChatGPT) because the study evaluates and analyzes ChatGPT&#x2019;s responses in direct comparison with human responses.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the Basic Science Research Program through the National Research Foundation of Korea (NRF) funded by the Ministry of Education, Science and Technology (RS-2023-NR077207).</p></sec><sec><title>Data Availability</title><p>The datasets generated or analyzed during this study are available in the GitHub repository [<xref ref-type="bibr" rid="ref29">29</xref>] and in the supplementary information files (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1</xref> and <xref ref-type="supplementary-material" rid="app3">3</xref>). The source dental chart images are not publicly available and cannot be shared owing to patient-privacy restrictions.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: AYC, DGS, JHK, JI, MJJ, SHE</p><p>Data curation: AYC, DGS, SHE</p><p>Formal analysis: SHE, JI, MJJ</p><p>Investigation: AYC, SHE, DGS, JI</p><p>Methodology: AYC, DGS, SHE</p><p>Project administration: JHK, JI, MJJ</p><p>Resources: DGS, SHE, JI, JHK</p><p>Software: SHE, JI, DGS</p><p>Supervision: DGS, JHK, JI, MJJ</p><p>Validation: JHK, JI, MJJ</p><p>Visualization: AYC, DGS, SHE</p><p>Writing &#x2013; original draft: AYC, SHE, DGS</p><p>Writing &#x2013; review &#x0026; editing: AYC, DGS, JHK, JI, MJJ, SHE</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>bidirectional encoder representations from transformers</p></def></def-item><def-item><term id="abb2">BLEU</term><def><p>bilingual evaluation understudy</p></def></def-item><def-item><term id="abb3">EM</term><def><p>exact match</p></def></def-item><def-item><term id="abb4">EMR</term><def><p>electronic medical record</p></def></def-item><def-item><term id="abb5">GUI</term><def><p>graphical user interface</p></def></def-item><def-item><term id="abb6">IRB</term><def><p>Institutional Review Board</p></def></def-item><def-item><term id="abb7">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb8">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb9">OCR</term><def><p>optical character recognition</p></def></def-item><def-item><term id="abb10">ROUGE-L</term><def><p>recall-oriented understudy for gisting evaluation-longest common subsequence</p></def></def-item><def-item><term id="abb11">SBERT</term><def><p>sentence bidirectional encoder representations from transformers</p></def></def-item><def-item><term id="abb12">SNUDH</term><def><p>Seoul National University Dental Hospital</p></def></def-item><def-item><term id="abb13">STROBE</term><def><p>Strengthening the Reporting of Observational Studies in Epidemiology</p></def></def-item><def-item><term id="abb14">USMLE</term><def><p>United States Medical Licensing Examination</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Merri&#x00EB;nboer</surname><given-names>JJG</given-names> </name><name name-style="western"><surname>Sweller</surname><given-names>J</given-names> </name></person-group><article-title>Cognitive load theory in health professional education: design principles and strategies</article-title><source>Med Educ</source><year>2010</year><month>01</month><volume>44</volume><issue>1</issue><fpage>85</fpage><lpage>93</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2923.2009.03498.x</pub-id><pub-id pub-id-type="medline">20078759</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Young</surname><given-names>JQ</given-names> </name><name name-style="western"><surname>Van Merrienboer</surname><given-names>J</given-names> </name><name name-style="western"><surname>Durning</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ten Cate</surname><given-names>O</given-names> </name></person-group><article-title>Cognitive load theory: implications for medical education: AMEE Guide no.86</article-title><source>Med Teach</source><year>2014</year><month>05</month><volume>36</volume><issue>5</issue><fpage>371</fpage><lpage>384</lpage><pub-id pub-id-type="doi">10.3109/0142159X.2014.889290</pub-id><pub-id pub-id-type="medline">24593808</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gordon</surname><given-names>M</given-names> </name><name name-style="western"><surname>Daniel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ajiboye</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A scoping review of artificial intelligence in medical education: BEME Guide no. 84</article-title><source>Med Teach</source><year>2024</year><month>04</month><volume>46</volume><issue>4</issue><fpage>446</fpage><lpage>470</lpage><pub-id pub-id-type="doi">10.1080/0142159X.2024.2314198</pub-id><pub-id pub-id-type="medline">38423127</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zitzmann</surname><given-names>NU</given-names> </name><name name-style="western"><surname>Matthisson</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ohla</surname><given-names>H</given-names> </name><name name-style="western"><surname>Joda</surname><given-names>T</given-names> </name></person-group><article-title>Digital undergraduate education in dentistry: a systematic review</article-title><source>Int J Environ Res Public Health</source><year>2020</year><month>05</month><day>7</day><volume>17</volume><issue>9</issue><fpage>3269</fpage><pub-id pub-id-type="doi">10.3390/ijerph17093269</pub-id><pub-id pub-id-type="medline">32392877</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levitin</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Grbic</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Finkelstein</surname><given-names>J</given-names> </name></person-group><article-title>Completeness of electronic dental records in a student clinic: retrospective analysis</article-title><source>JMIR Med Inform</source><year>2019</year><month>03</month><day>21</day><volume>7</volume><issue>1</issue><fpage>e13008</fpage><pub-id pub-id-type="doi">10.2196/13008</pub-id><pub-id pub-id-type="medline">30896435</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Acharya</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schroeder</surname><given-names>D</given-names> </name><name name-style="western"><surname>Schwei</surname><given-names>K</given-names> </name><name name-style="western"><surname>Chyou</surname><given-names>PH</given-names> </name></person-group><article-title>Update on electronic dental record and clinical computing adoption among dental practices in the United States</article-title><source>Clin Med Res</source><year>2017</year><month>12</month><volume>15</volume><issue>3-4</issue><fpage>59</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.3121/cmr.2017.1380</pub-id><pub-id pub-id-type="medline">29229631</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gilson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Safranek</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>How does ChatGPT perform on the United States Medical Licensing Examination (USMLE)? The implications of large language models for medical education and knowledge assessment</article-title><source>JMIR Med Educ</source><year>2023</year><month>02</month><day>8</day><volume>9</volume><fpage>e45312</fpage><pub-id pub-id-type="doi">10.2196/45312</pub-id><pub-id pub-id-type="medline">36753318</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kung</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Cheatham</surname><given-names>M</given-names> </name><name name-style="western"><surname>Medenilla</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title><source>PLoS Digit Health</source><year>2023</year><month>02</month><volume>2</volume><issue>2</issue><fpage>e0000198</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id><pub-id pub-id-type="medline">36812645</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eysenbach</surname><given-names>G</given-names> </name></person-group><article-title>The role of ChatGPT, generative language models, and artificial intelligence in medical education: a conversation with ChatGPT and a call for papers</article-title><source>JMIR Med Educ</source><year>2023</year><month>03</month><day>6</day><volume>9</volume><fpage>e46885</fpage><pub-id pub-id-type="doi">10.2196/46885</pub-id><pub-id pub-id-type="medline">36863937</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shorey</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mattar</surname><given-names>C</given-names> </name><name name-style="western"><surname>Pereira</surname><given-names>TLB</given-names> </name><name name-style="western"><surname>Choolani</surname><given-names>M</given-names> </name></person-group><article-title>A scoping review of ChatGPT&#x2019;s role in healthcare education and research</article-title><source>Nurse Educ Today</source><year>2024</year><month>04</month><volume>135</volume><fpage>106121</fpage><pub-id pub-id-type="doi">10.1016/j.nedt.2024.106121</pub-id><pub-id pub-id-type="medline">38340639</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cheng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>L</given-names> </name></person-group><article-title>A review of ChatGPT in medical education: exploring advantages and limitations</article-title><source>Int J Surg</source><year>2025</year><month>07</month><day>1</day><volume>111</volume><issue>7</issue><fpage>4586</fpage><lpage>4602</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000002505</pub-id><pub-id pub-id-type="medline">40465793</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ji</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>N</given-names> </name><name name-style="western"><surname>Frieske</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Survey of hallucination in natural language generation</article-title><source>ACM Comput Surv</source><year>2023</year><month>12</month><day>31</day><volume>55</volume><issue>12</issue><fpage>1</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1145/3571730</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Hond</surname><given-names>A</given-names> </name><name name-style="western"><surname>Leeuwenberg</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bartels</surname><given-names>R</given-names> </name><etal/></person-group><article-title>From text to treatment: the crucial role of validation for generative large language models in health care</article-title><source>Lancet Digit Health</source><year>2024</year><month>07</month><volume>6</volume><issue>7</issue><fpage>e441</fpage><lpage>e443</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(24)00111-0</pub-id><pub-id pub-id-type="medline">38906607</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Benoit</surname><given-names>B</given-names> </name><name name-style="western"><surname>Fr&#x00E9;d&#x00E9;ric</surname><given-names>B</given-names> </name><name name-style="western"><surname>Jean-Charles</surname><given-names>D</given-names> </name></person-group><article-title>Current state of dental informatics in the field of health information systems: a scoping review</article-title><source>BMC Oral Health</source><year>2022</year><month>04</month><day>19</day><volume>22</volume><issue>1</issue><fpage>131</fpage><pub-id pub-id-type="doi">10.1186/s12903-022-02163-9</pub-id><pub-id pub-id-type="medline">35439988</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kavadella</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dias da Silva</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Kaklamanos</surname><given-names>EG</given-names> </name><name name-style="western"><surname>Stamatopoulos</surname><given-names>V</given-names> </name><name name-style="western"><surname>Giannakopoulos</surname><given-names>K</given-names> </name></person-group><article-title>Evaluation of ChatGPT&#x2019;s real-life implementation in undergraduate dental education: mixed methods study</article-title><source>JMIR Med Educ</source><year>2024</year><month>01</month><day>31</day><volume>10</volume><fpage>e51344</fpage><pub-id pub-id-type="doi">10.2196/51344</pub-id><pub-id pub-id-type="medline">38111256</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abril-Gonzalez</surname><given-names>M</given-names> </name><name name-style="western"><surname>Portilla</surname><given-names>FA</given-names> </name><name name-style="western"><surname>Jaramillo-Mejia</surname><given-names>MC</given-names> </name></person-group><article-title>Standard health level seven for odontological digital imaging</article-title><source>Telemed J E Health</source><year>2017</year><month>01</month><volume>23</volume><issue>1</issue><fpage>63</fpage><lpage>70</lpage><pub-id pub-id-type="doi">10.1089/tmj.2015.0251</pub-id><pub-id pub-id-type="medline">27248059</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Giannakopoulos</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kavadella</surname><given-names>A</given-names> </name><name name-style="western"><surname>Aaqel Salim</surname><given-names>A</given-names> </name><name name-style="western"><surname>Stamatopoulos</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kaklamanos</surname><given-names>EG</given-names> </name></person-group><article-title>Evaluation of the performance of generative AI large language models ChatGPT, Google Bard, and Microsoft Bing Chat in supporting evidence-based dentistry: comparative mixed methods study</article-title><source>J Med Internet Res</source><year>2023</year><month>12</month><day>28</day><volume>25</volume><fpage>e51580</fpage><pub-id pub-id-type="doi">10.2196/51580</pub-id><pub-id pub-id-type="medline">38009003</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adams</surname><given-names>NE</given-names> </name></person-group><article-title>Bloom&#x2019;s taxonomy of cognitive learning objectives</article-title><source>J Med Libr Assoc</source><year>2015</year><month>07</month><volume>103</volume><issue>3</issue><fpage>152</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.3163/1536-5050.103.3.010</pub-id><pub-id pub-id-type="medline">26213509</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pampari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Raghavan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>J</given-names> </name></person-group><article-title>emrQA: a large corpus for question answering on electronic medical records</article-title><year>2018</year><access-date>2026-08-05</access-date><conf-name>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Oct 31 to Nov 4, 2018</conf-date><conf-loc>Brussels, Belgium</conf-loc><fpage>2357</fpage><lpage>2368</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://aclweb.org/anthology/D18-1">http://aclweb.org/anthology/D18-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D18-1258</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raza</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Venkatesh</surname><given-names>KP</given-names> </name><name name-style="western"><surname>Kvedar</surname><given-names>JC</given-names> </name></person-group><article-title>Generative AI and large language models in health care: pathways to implementation</article-title><source>NPJ Digit Med</source><year>2024</year><month>03</month><day>7</day><volume>7</volume><issue>1</issue><fpage>62</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00988-4</pub-id><pub-id pub-id-type="medline">38454007</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arora</surname><given-names>A</given-names> </name><name name-style="western"><surname>Arora</surname><given-names>A</given-names> </name></person-group><article-title>The promise of large language models in health care</article-title><source>Lancet</source><year>2023</year><month>02</month><day>25</day><volume>401</volume><issue>10377</issue><fpage>641</fpage><pub-id pub-id-type="doi">10.1016/S0140-6736(23)00216-7</pub-id><pub-id pub-id-type="medline">36841609</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Papineni</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roukos</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>WJ</given-names> </name></person-group><article-title>BLEU: a method for automatic evaluation of machine translation</article-title><year>2002</year><conf-name>40th Annual Meeting of the Association for Computational Linguistics (ACL)</conf-name><conf-date>Jul 6-12, 2002</conf-date><conf-loc>Philadelphia, PA</conf-loc><fpage>311</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.3115/1073083.1073135</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Denkowski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lavie</surname><given-names>A</given-names> </name></person-group><article-title>Meteor universal: language specific translation evaluation for any target language</article-title><year>2014</year><access-date>2026-08-05</access-date><conf-name>Proceedings of the Ninth Workshop on Statistical Machine Translation</conf-name><conf-date>Jun 26-27, 2014</conf-date><conf-loc>Baltimore, Maryland</conf-loc><fpage>376</fpage><lpage>380</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://aclweb.org/anthology/W14-33">http://aclweb.org/anthology/W14-33</ext-link></comment><pub-id pub-id-type="doi">10.3115/v1/W14-3348</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Post</surname><given-names>M</given-names> </name></person-group><article-title>A call for clarity in reporting BLEU scores</article-title><year>2018</year><access-date>2026-08-05</access-date><conf-name>Proceedings of the Third Conference on Machine Translation</conf-name><conf-date>Oct 31 to Nov 1, 2018</conf-date><conf-loc>Belgium, Brussels</conf-loc><fpage>186</fpage><lpage>191</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://aclweb.org/anthology/W18-63">http://aclweb.org/anthology/W18-63</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/W18-6319</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name></person-group><article-title>ROUGE: a package for automatic evaluation of summaries</article-title><access-date>2026-08-05</access-date><conf-name>Text Summarization Branches Out: Proceedings of the ACL-04 Workshop</conf-name><conf-date>Jul 25-26, 2004</conf-date><conf-loc>Barcelona, Spain</conf-loc><fpage>74</fpage><lpage>81</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W04-1013/">https://aclanthology.org/W04-1013/</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kishore</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Weinberger</surname><given-names>KQ</given-names> </name><name name-style="western"><surname>Artzi</surname><given-names>Y</given-names> </name></person-group><article-title>BERTScore: Evaluating text generation with BERT</article-title><access-date>2026-08-19</access-date><conf-name>2020 International Conference on Learning Representations</conf-name><conf-date>Apr 27-30, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=SkeHuCVFDr">https://openreview.net/pdf?id=SkeHuCVFDr</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name></person-group><article-title>Sentence-BERT: sentence embeddings using siamese BERT-networks</article-title><year>2019</year><access-date>2026-08-05</access-date><conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-date>Nov 3-7, 2019</conf-date><conf-loc>Hong Kong, China</conf-loc><fpage>3982</fpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.aclweb.org/anthology/D19-1">https://www.aclweb.org/anthology/D19-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D19-1410</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Eo</surname><given-names>SH</given-names> </name></person-group><article-title>Benchmarking multimodal LLMs on dental chart image interpretation: a comparison with students and clinicians</article-title><source>GitHub</source><access-date>2026-06-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/sooheang/eval-genai-dentistry">https://github.com/sooheang/eval-genai-dentistry</ext-link></comment></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nam</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>DY</given-names> </name><name name-style="western"><surname>Kyung</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Multimodal large language models in medical imaging: current state and future directions</article-title><source>Korean J Radiol</source><year>2025</year><month>10</month><volume>26</volume><issue>10</issue><fpage>900</fpage><lpage>923</lpage><pub-id pub-id-type="doi">10.3348/kjr.2025.0599</pub-id><pub-id pub-id-type="medline">41015856</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bradshaw</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Tie</surname><given-names>X</given-names> </name><name name-style="western"><surname>Warner</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name></person-group><article-title>Large language models and large multimodal models in medical imaging: a primer for physicians</article-title><source>J Nucl Med</source><year>2025</year><month>02</month><day>3</day><volume>66</volume><issue>2</issue><fpage>173</fpage><lpage>182</lpage><pub-id pub-id-type="doi">10.2967/jnumed.124.268072</pub-id><pub-id pub-id-type="medline">39819692</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>CF</given-names> </name><etal/></person-group><article-title>Towards a holistic framework for multimodal LLM in 3D brain CT radiology report generation</article-title><source>Nat Commun</source><year>2025</year><month>03</month><day>6</day><volume>16</volume><issue>1</issue><fpage>2258</fpage><pub-id pub-id-type="doi">10.1038/s41467-025-57426-0</pub-id><pub-id pub-id-type="medline">40050277</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yi</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>T</given-names> </name><name name-style="western"><surname>Albert</surname><given-names>MV</given-names> </name></person-group><article-title>A survey on multimodal large language models in radiology for report generation and visual question answering</article-title><source>Information</source><year>2025</year><volume>16</volume><issue>2</issue><fpage>136</fpage><pub-id pub-id-type="doi">10.3390/info16020136</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Strasser</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Anschuetz</surname><given-names>W</given-names> </name><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hastings</surname><given-names>J</given-names> </name></person-group><article-title>Performance evaluation of large language models in multilingual medical multiple-choice questions: mixed methods study</article-title><source>JMIR Med Educ</source><year>2026</year><month>03</month><day>5</day><volume>12</volume><fpage>e81399</fpage><pub-id pub-id-type="doi">10.2196/81399</pub-id><pub-id pub-id-type="medline">41813244</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Croxford</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pellegrino</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Current and future state of evaluation of large language models for medical summarization tasks</article-title><source>NPJ Health Syst</source><year>2025</year><volume>2</volume><issue>1</issue><fpage>6</fpage><pub-id pub-id-type="doi">10.1038/s44401-024-00011-2</pub-id><pub-id pub-id-type="medline">40124388</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aster</surname><given-names>A</given-names> </name><name name-style="western"><surname>Laupichler</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Rockwell-Kollmann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Masala</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bala</surname><given-names>E</given-names> </name><name name-style="western"><surname>Raupach</surname><given-names>T</given-names> </name></person-group><article-title>ChatGPT and other large language models in medical education &#x2014; scoping literature review</article-title><source>MedSciEduc</source><year>2025</year><month>02</month><volume>35</volume><issue>1</issue><fpage>555</fpage><lpage>567</lpage><pub-id pub-id-type="doi">10.1007/s40670-024-02206-6</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Roustan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bastardot</surname><given-names>F</given-names> </name></person-group><article-title>The clinicians&#x2019; guide to large language models: a general perspective with a focus on hallucinations</article-title><source>Interact J Med Res</source><year>2025</year><month>01</month><day>28</day><volume>14</volume><fpage>e59823</fpage><pub-id pub-id-type="doi">10.2196/59823</pub-id><pub-id pub-id-type="medline">39874574</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lucas</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Upperman</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>JR</given-names> </name></person-group><article-title>A systematic review of large language models and their implications in medical education</article-title><source>Med Educ</source><year>2024</year><month>11</month><volume>58</volume><issue>11</issue><fpage>1276</fpage><lpage>1285</lpage><pub-id pub-id-type="doi">10.1111/medu.15402</pub-id><pub-id pub-id-type="medline">38639098</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="web"><article-title>OpenAI</article-title><source>GPT-5 system card</source><year>2025</year><access-date>2025-10-11</access-date><publisher-name>OpenAI</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://cdn.openai.com/gpt-5-system-card.pdf">https://cdn.openai.com/gpt-5-system-card.pdf</ext-link></comment></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rao</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Esmail</surname><given-names>KP</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>RS</given-names> </name><etal/></person-group><article-title>Large language model performance and clinical reasoning tasks</article-title><source>JAMA Netw Open</source><year>2026</year><month>04</month><day>1</day><volume>9</volume><issue>4</issue><fpage>e264003</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2026.4003</pub-id><pub-id pub-id-type="medline">41973425</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>W</given-names> </name><etal/></person-group><article-title>A survey on hallucination in large language models: principles, taxonomy, challenges, and open questions</article-title><source>ACM Trans Inf Syst</source><year>2025</year><month>03</month><day>31</day><volume>43</volume><issue>2</issue><fpage>1</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.1145/3703155</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Howe</surname><given-names>B</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>LL</given-names> </name></person-group><article-title>Characterizing LLM abstention behavior in science QA with context perturbations</article-title><access-date>2026-08-06</access-date><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Nov 12-16, 2024</conf-date><conf-loc>Miami, FL</conf-loc><fpage>3437</fpage><lpage>3450</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.findings-emnlp">https://aclanthology.org/2024.findings-emnlp</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2024.findings-emnlp.197</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>M</given-names> </name></person-group><article-title>A mathematical investigation of hallucination and creativity in GPT models</article-title><source>Mathematics</source><year>2023</year><volume>11</volume><issue>10</issue><fpage>2320</fpage><pub-id pub-id-type="doi">10.3390/math11102320</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Asgari</surname><given-names>E</given-names> </name><name name-style="western"><surname>Monta&#x00F1;a-Brown</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dubois</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>13</day><volume>8</volume><issue>1</issue><fpage>274</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01670-7</pub-id><pub-id pub-id-type="medline">40360677</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>JD</given-names> </name><etal/></person-group><article-title>Multi-model assurance analysis showing large language models are highly vulnerable to adversarial hallucination attacks during clinical decision support</article-title><source>Commun Med (Lond)</source><year>2025</year><month>08</month><day>2</day><volume>5</volume><issue>1</issue><fpage>330</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-01021-3</pub-id><pub-id pub-id-type="medline">40753316</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ullah</surname><given-names>R</given-names> </name><name name-style="western"><surname>Shaikh</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Shahani</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lone</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Fareed</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Zafar</surname><given-names>MS</given-names> </name></person-group><article-title>Comparing ChatGPT and dental students&#x2019; performance in an introduction to dental anatomy examination: a cross-sectional study</article-title><source>Eur J Dent</source><year>2026</year><month>02</month><volume>20</volume><issue>1</issue><fpage>287</fpage><lpage>294</lpage><pub-id pub-id-type="doi">10.1055/s-0045-1808254</pub-id><pub-id pub-id-type="medline">40359999</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lyell</surname><given-names>D</given-names> </name><name name-style="western"><surname>Coiera</surname><given-names>E</given-names> </name></person-group><article-title>Automation bias and verification complexity: a systematic review</article-title><source>J Am Med Inform Assoc</source><year>2017</year><month>03</month><day>1</day><volume>24</volume><issue>2</issue><fpage>423</fpage><lpage>431</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocw105</pub-id><pub-id pub-id-type="medline">27516495</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hattie</surname><given-names>J</given-names> </name><name name-style="western"><surname>Timperley</surname><given-names>H</given-names> </name></person-group><article-title>The power of feedback</article-title><source>Rev Educ Res</source><year>2007</year><month>03</month><volume>77</volume><issue>1</issue><fpage>81</fpage><lpage>112</lpage><pub-id pub-id-type="doi">10.3102/003465430298487</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Claman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sezgin</surname><given-names>E</given-names> </name></person-group><article-title>Artificial intelligence in dental education: opportunities and challenges of large language models and multimodal foundation models</article-title><source>JMIR Med Educ</source><year>2024</year><month>09</month><day>27</day><volume>10</volume><fpage>e52346</fpage><pub-id pub-id-type="doi">10.2196/52346</pub-id><pub-id pub-id-type="medline">39331527</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Masters</surname><given-names>K</given-names> </name></person-group><article-title>Ethical use of artificial intelligence in health professions education: AMEE Guide no. 158</article-title><source>Med Teach</source><year>2023</year><month>06</month><volume>45</volume><issue>6</issue><fpage>574</fpage><lpage>584</lpage><pub-id pub-id-type="doi">10.1080/0142159X.2023.2186203</pub-id><pub-id pub-id-type="medline">36912253</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pham</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Karunaratne</surname><given-names>N</given-names> </name><name name-style="western"><surname>Exintaris</surname><given-names>B</given-names> </name><etal/></person-group><article-title>The impact of generative AI on health professional education: a systematic review in the context of student learning</article-title><source>Med Educ</source><year>2025</year><month>12</month><volume>59</volume><issue>12</issue><fpage>1280</fpage><lpage>1289</lpage><pub-id pub-id-type="doi">10.1111/medu.15746</pub-id><pub-id pub-id-type="medline">40533396</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Masters</surname><given-names>K</given-names> </name></person-group><article-title>Artificial intelligence in medical education</article-title><source>Med Teach</source><year>2019</year><month>09</month><volume>41</volume><issue>9</issue><fpage>976</fpage><lpage>980</lpage><pub-id pub-id-type="doi">10.1080/0142159X.2019.1595557</pub-id><pub-id pub-id-type="medline">31007106</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Izquierdo-Condoy</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Arias-Intriago</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tello-De-la-Torre</surname><given-names>A</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ortiz-Prado</surname><given-names>E</given-names> </name></person-group><article-title>Generative artificial intelligence in medical education: enhancing critical thinking or undermining cognitive autonomy?</article-title><source>J Med Internet Res</source><year>2025</year><month>11</month><day>3</day><volume>27</volume><fpage>e76340</fpage><pub-id pub-id-type="doi">10.2196/76340</pub-id><pub-id pub-id-type="medline">41183320</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Akinci D&#x2019;Antonoli</surname><given-names>T</given-names> </name><name name-style="western"><surname>Stanzione</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bluethgen</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Large language models in radiology: fundamentals, applications, ethical considerations, risks, and future directions</article-title><source>Diagn Interv Radiol</source><year>2024</year><month>03</month><day>6</day><volume>30</volume><issue>2</issue><fpage>80</fpage><lpage>90</lpage><pub-id pub-id-type="doi">10.4274/dir.2023.232417</pub-id><pub-id pub-id-type="medline">37789676</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Keshavarz</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bagherieh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Nabipoorashrafi</surname><given-names>SA</given-names> </name><etal/></person-group><article-title>ChatGPT in radiology: a systematic review of performance, pitfalls, and future perspectives</article-title><source>Diagn Interv Imaging</source><year>2024</year><volume>105</volume><issue>7-8</issue><fpage>251</fpage><lpage>265</lpage><pub-id pub-id-type="doi">10.1016/j.diii.2024.04.003</pub-id><pub-id pub-id-type="medline">38679540</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harigai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Toyama</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Nagano</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Response accuracy of GPT-4 across languages: insights from an expert-level diagnostic radiology examination in Japan</article-title><source>Jpn J Radiol</source><year>2025</year><month>02</month><volume>43</volume><issue>2</issue><fpage>319</fpage><lpage>329</lpage><pub-id pub-id-type="doi">10.1007/s11604-024-01673-6</pub-id><pub-id pub-id-type="medline">39466356</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>ES</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>GH</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SP</given-names> </name></person-group><article-title>Evaluation of GPT-4o and Gemini Advanced on the Korean National Dental Licensing examination: accuracy, consistency, and question generation</article-title><source>J Dent Sci</source><year>2026</year><month>01</month><volume>21</volume><issue>1</issue><fpage>96</fpage><lpage>102</lpage><pub-id pub-id-type="doi">10.1016/j.jds.2025.07.020</pub-id><pub-id pub-id-type="medline">41585183</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bereuter</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Geissler</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Klimova</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Benchmarking vision capabilities of large language models in surgical examination questions</article-title><source>J Surg Educ</source><year>2025</year><month>04</month><volume>82</volume><issue>4</issue><fpage>103442</fpage><pub-id pub-id-type="doi">10.1016/j.jsurg.2025.103442</pub-id><pub-id pub-id-type="medline">39923296</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wieling</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rawee</surname><given-names>J</given-names> </name><name name-style="western"><surname>van Noord</surname><given-names>G</given-names> </name></person-group><article-title>Reproducibility in computational linguistics: are we willing to share?</article-title><source>Comput Linguist</source><year>2018</year><month>12</month><volume>44</volume><issue>4</issue><fpage>641</fpage><lpage>649</lpage><pub-id pub-id-type="doi">10.1162/coli_a_00330</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zaharia</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>J</given-names> </name></person-group><article-title>How Is ChatGPT&#x2019;s behavior changing over time?</article-title><source>Harvard Data Sci Rev</source><year>2024</year><volume>6</volume><issue>2</issue><pub-id pub-id-type="doi">10.1162/99608f92.5317da47</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Dataset of 360 case-specific questions (in both the original Korean and translated English), human reference answers, raw outputs from the 9 evaluated models, and direct URL links to the original conversational logs.</p><media xlink:href="mededu_v12i1e91809_app1.xlsx" xlink:title="XLSX File, 272 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Supplementary tables including exploratory qualitative feedback on educational utility, interrater reliability of the human reference standard, robustness of the evaluation metrics, model safety (fabrication rates), effect sizes for human-model comparisons, and expert adjudication of clinically consequential discrepancies.</p><media xlink:href="mededu_v12i1e91809_app2.docx" xlink:title="DOCX File, 23 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Complete per-item metric evaluation results across 360 questions, comparing each response with the gold-standard references using 7 metrics (Exact Match, <italic>F</italic><sub>1</sub>-score, BLEU, SacreBLEU, ROUGE-L, BERTScore, and SBERT). The dataset includes results for all evaluated models as well as student and resident raters.</p><media xlink:href="mededu_v12i1e91809_app3.xlsx" xlink:title="XLSX File, 363 KB"/></supplementary-material><supplementary-material id="app4"><label>Checklist 1</label><p>STROBE checklist.</p><media xlink:href="mededu_v12i1e91809_app4.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material></app-group></back></article>