<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id><journal-id journal-id-type="publisher-id">mededu</journal-id><journal-id journal-id-type="index">20</journal-id><journal-title>JMIR Medical Education</journal-title><abbrev-journal-title>JMIR Med Educ</abbrev-journal-title><issn pub-type="epub">2369-3762</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v12i1e70199</article-id><article-id pub-id-type="doi">10.2196/70199</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Performance of GPT-4o and Claude in Medical Ethics Scenarios: Comparative Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Desai</surname><given-names>Karishma R</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gorsky</surname><given-names>Anna L</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Zelenski</surname><given-names>Nicole A</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Division of Hand Surgery, Department of Orthopaedic Surgery, Emory University</institution><addr-line>80 Jesse Hill Jr Drive SE</addr-line><addr-line>Atlanta</addr-line><addr-line>GA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Lesselroth</surname><given-names>Blake</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Leung</surname><given-names>Fok-Han</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Causio</surname><given-names>Francesco Andrea</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Donner</surname><given-names>Sascha</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Nicole A Zelenski, MD, Division of Hand Surgery, Department of Orthopaedic Surgery, Emory University, 80 Jesse Hill Jr Drive SE, Atlanta, GA, 30303, United States, 1 404-778-1550; <email>nicole.ann.zelenski@emory.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>23</day><month>7</month><year>2026</year></pub-date><volume>12</volume><elocation-id>e70199</elocation-id><history><date date-type="received"><day>17</day><month>12</month><year>2024</year></date><date date-type="rev-recd"><day>12</day><month>05</month><year>2026</year></date><date date-type="accepted"><day>15</day><month>05</month><year>2026</year></date></history><copyright-statement>&#x00A9; Karishma R Desai, Anna L Gorsky, Nicole A Zelenski. Originally published in JMIR Medical Education (<ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org">https://mededu.jmir.org</ext-link>), 23.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org/">https://mededu.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mededu.jmir.org/2026/1/e70199"/><abstract><sec><title>Background</title><p>The emergence of AI technology has sparked curiosity regarding the capabilities of large language models (LLMs) in the field of medicine. Minimal research exists regarding the proficiency of various AI models in ethics scenarios, specifically in specialty-based scenarios.</p></sec><sec><title>Objective</title><p>This study aimed to compare the performance of GPT-4o and Claude Sonnet 4 on ethics questions with that of medical students and orthopedic residents.</p></sec><sec sec-type="methods"><title>Methods</title><p>A total of 200 ethical or legal scenario questions were randomly selected from question banks targeted for third- and fourth-year medical students (UWorld, AMBOSS) and orthopedic residents (OrthoBullets). Questions at the medical student level were exclusively text-based, while resident-level questions included text-based questions accompanied by images. Each question was entered identically into each AI model 3 separate times. If answers varied between trials, the answer provided most frequently by the model was used as the selected answer.</p></sec><sec sec-type="results"><title>Results</title><p>GPT-4o correctly answered 140 (70%) of 200 questions, which was similar to the average human test taker score of 71% (~142/200 questions). Claude correctly answered 180 (89%) questions, a score greater than that of human test takers and significantly better than GPT-4o (<italic>P&#x003C;</italic>.001). Claude scored significantly higher than GPT-4o in almost all question categories. GPT-4o provided different responses to identically worded trials for 27 (21%) of 130 general questions and 3 (4%) of 70 orthopedic questions (<italic>P=</italic>.002), while Claude did not have a significant difference in variability between these 2 groups (general: 16/130, 12% vs orthopedic: 3/70, 4%; <italic>P</italic>=.06). GPT-4o selected the incorrect response for 60 (30%) total questions and chose the incorrect response most commonly selected by humans significantly more frequently on UWorld interpersonal-specific questions (30/40, 75%) than on UWorld all social sciences (27/40, 68%; <italic>P</italic>=.03). Claude showed no significant difference in the rate of most common incorrect response selection between question categories.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>These results suggest that GPT-4o can potentially answer both general and specialty-specific ethical questions with similar proficiency to sample groups of both medical students and orthopedic residents, while Claude AI performs significantly better than both humans and GPT-4o. Variables such as AI model framework and training data may drive the observed difference in performance, but the exact cause cannot be definitively isolated without intentional testing. Therefore, further research is needed to ensure safety by minimizing output variability before integrating AI as a patient-facing resource.</p></sec></abstract><kwd-group><kwd>medical education</kwd><kwd>ethics</kwd><kwd>legal</kwd><kwd>ChatGPT</kwd><kwd>artificial intelligence</kwd><kwd>medical student</kwd><kwd>orthopedics</kwd><kwd>Claude</kwd><kwd>resident</kwd><kwd>health</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Interpersonal skills are an invaluable necessity for health care providers. Impactful medical practice requires the ability to adapt treatment plans to a patient&#x2019;s unique preferences, which is founded on the development of the patient-provider relationship. Numerous studies have shown that effective physician communication is positively correlated with patient satisfaction [<xref ref-type="bibr" rid="ref1">1</xref>] and positive health outcomes [<xref ref-type="bibr" rid="ref2">2</xref>]. Patient involvement in their own medical care, often termed &#x201C;shared decision-making,&#x201D; has also been observed to correlate with increased overall and disease-specific quality of life scores in orthopedic patients [<xref ref-type="bibr" rid="ref3">3</xref>]. The now near-ubiquity of internet usage has resulted in patients frequently consulting online materials for additional medical information. Nearly all American households had internet access in 2021 [<xref ref-type="bibr" rid="ref4">4</xref>], and recent Centers for Disease Control and Prevention data suggest that more than half of adults (58.5%) use the internet for medical advice [<xref ref-type="bibr" rid="ref5">5</xref>]. This trend is also mirrored internationally, with 1 study showing that 64.3% of Lebanese patients consulted web resources for acute symptoms, with 19.2% seeking web information before physician consultation [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>The increasing popularity of AI technologies has introduced a new potential source for internet-accessible medical information. Large language models (LLMs), such as GPT-4o (OpenAI), are currently being evaluated for their ability to play a role in health care. AI has been shown to accurately process medical knowledge and show proficiency at the level of third- and fourth-year medical students in preparation for the United States Medical Licensing Examination [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. In addition, GPT-4o has been challenged with postgraduate examination materials, including preparation materials for the American Board of Surgery In-Training Examination [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>], neurology specialist examinations [<xref ref-type="bibr" rid="ref12">12</xref>], and the Orthopaedic In-Training Examination [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>] and, in most cases, has performed at the level of a human medical trainee. Recent exploration has shown that ChatGPT models can also directly respond to patient questions about common conditions across a variety of specialties with generally satisfactory answers [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Nevertheless, integration of AI into health care settings remains controversial due to a lack of clarity surrounding its strengths and limitations, and it is unclear whether AI can be safely recommended to serve as a substitute or even as an adjunct to physician consultation [<xref ref-type="bibr" rid="ref18">18</xref>]. The &#x201C;soft skills&#x201D; of medical training include compassion and empathetic communication, which are essential to identifying and navigating the nuances present within each individual patient interaction. Although AI technologies have shown some capability in correctly reproducing technical knowledge and answering patient questions, their capability to identify and process user emotion is not well understood.</p><p>LLMs are modeled on neural networks, in which units or &#x201C;tokens&#x201D; of data are connected and linked by associations, which are strengthened or weakened based on the presence of patterns within training data [<xref ref-type="bibr" rid="ref19">19</xref>]. Essentially, the models use the patterns identified from their training data to generate outputs that are the most probabilistically likely &#x201C;correct&#x201D; responses [<xref ref-type="bibr" rid="ref20">20</xref>]. Transformer architecture is the modern evolution of neural networks and uses &#x201C;self-attention&#x201D; to consider and weight tokens in relation to each other [<xref ref-type="bibr" rid="ref21">21</xref>]. Through the training process, the model adapts the level of attention given to various tokens to identify the most relevant components of the input and better generate more relevant responses in the given context [<xref ref-type="bibr" rid="ref21">21</xref>]. Ultimately, transformer architecture improves the statistical predictive accuracy of LLMs, especially in the processing of large volumes of inputs, by more accurately weighing the relevance of input tokens [<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>Training can occur through the establishment of parameters, which guide the model in processing data inputs to promote or oppose specific outputs; positive reinforcement of a specific association teaches the model that the reinforced logic has an increased likelihood of leading to the correct output [<xref ref-type="bibr" rid="ref22">22</xref>]. Feedback can be provided to the model based on the final output (outcome supervision) or throughout the logical process (process supervision), and the model applies this feedback and recalculated probabilities to direct the processing of new data [<xref ref-type="bibr" rid="ref23">23</xref>]. Therefore, the veracity of AI-generated outputs is dependent directly on the quality and accuracy of the training data and subtleties within the inputs themselves, rather than solely on known truths [<xref ref-type="bibr" rid="ref24">24</xref>]. As a result, AI models have been documented to &#x201C;hallucinate&#x201D; or fabricate information and present it convincingly as fact [<xref ref-type="bibr" rid="ref25">25</xref>]. Researchers in one study found that a number of inconsistencies and hallucinations were noted when questions on sensitive or obscure topics were presented to several AI models [<xref ref-type="bibr" rid="ref26">26</xref>], supporting the hypothesis that AI performance declines when available data are limited. &#x201C;Factuality&#x201D; hallucinations, when models report information that violates known facts, and &#x201C;faithfulness&#x201D; hallucinations, when models deviate from the information provided in the given input, are theorized to occur due to inconsistent training and model evaluation metrics [<xref ref-type="bibr" rid="ref27">27</xref>].</p><p>Despite numerous studies evaluating AI&#x2019;s accuracy in answering questions based on empiric knowledge, the literature supporting AI&#x2019;s performance in health care&#x2013;based medical ethical, legal, and social scenarios is limited and inconsistent. Some reports demonstrate that LLMs can answer ethical scenario questions [<xref ref-type="bibr" rid="ref28">28</xref>] and respond to patient clinical questions with answers that are rated positively for empathy by human raters [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>], while others found ChatGPT models performed worse on ethics questions compared to medical trainees [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. When subjected to emotion-related testing, AI demonstrated the ability to correctly identify and manage emotions but struggled when associating emotions to rationale and scored lower than the human average on the &#x201C;using emotions to facilitate thought&#x201D; section of the examination [<xref ref-type="bibr" rid="ref33">33</xref>].</p><p>Variant AI models have been designed to address the general safety and ethical concerns surrounding the outputs produced by LLMs. One such model is Claude, created by Anthropic, which describes it as &#x201C;helpful, honest, and harmless&#x201D; [<xref ref-type="bibr" rid="ref34">34</xref>]. Claude was developed using a novel &#x201C;Constitutional AI&#x201D; method, meaning that instead of using human feedback during the training process, feedback is instead provided by the AI model itself, guided by a provided set of &#x201C;constitutional principles&#x201D; [<xref ref-type="bibr" rid="ref35">35</xref>]. During the supervised training process, the model is asked prompts that will purposefully generate a &#x201C;harmful&#x201D; output, for example, asking for instructions on how to break into a house. The model is then asked to critique the output based on a constitutional principle (&#x201C;how is the last response unethical or illegal&#x201D;) and then to revise the output according to the critique (&#x201C;rewrite the response to remove unethical or illegal content&#x201D;) [<xref ref-type="bibr" rid="ref36">36</xref>]. In the reinforcement learning or unsupervised phase, the model is asked to generate multiple responses to a prompt, and a different feedback model is asked to choose one of the responses also based on a constitutional principle (&#x201C;which of these responses is the most ethical and legal?&#x201D;) [<xref ref-type="bibr" rid="ref36">36</xref>]. Therefore, humans are not providing feedback for each model response themselves, but instead instructing a &#x201C;teaching&#x201D; AI to guide the &#x201C;learning&#x201D; model to align with specific rules, or constitutional principles. Anthropic sources its constitutional principles from a number of ethic-centered documents, including the UN Declaration of Human Rights, as well as principles developed by other AI laboratories and Anthropic researchers themselves during model training to discourage harm, aggression, offense, violence, and so on [<xref ref-type="bibr" rid="ref35">35</xref>]. Although Claude is intentionally designed with ethics and safety in mind, there are no available studies specifically exploring Claude&#x2019;s performances in ethics-related health care scenarios.</p><p>Throughout medical school and postgraduate medical training, learners are taught how to manage complex social issues and interpersonal interactions with a high degree of emotional intelligence and empathy. Even ethical dilemmas, which are addressed using well-defined ethical principles, must be navigated within the larger context of patient emotion. Medical training examinations often use questions about end-of-life discussions, patient safety, communication strategies, and informed consent to assess these intangible skills. This study aimed to evaluate the ability of GPT-4o and Claude Sonnet 4 to answer health care&#x2013;based ethical medical board&#x2013;style questions accurately and consistently. We hypothesized that GPT-4o and Claude will differ in performance between question levels and categories.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><p>All questions used for testing were scenario-based multiple-choice questions for which there was only one correct answer. Medical student&#x2013;level questions were sourced from the UWorld Step 2 CK question bank and AMBOSS, 2 resources designed to prepare third- and fourth-year medical students for the Step 2 benchmark examination. The UWorld system allows for question sourcing from selected subject categories. One category of interest was the broad &#x201C;Social Sciences (Ethics/Legal/Professional)&#x201D; category, which involves the subcategories of Communication and Interpersonal Skills, Healthcare Policy and Economics, Medical Ethics and Jurisprudence, Patient Safety, and System-based practice and Quality Improvement. The subcategory &#x201C;Communication and Interpersonal Skills&#x201D; was also specifically selected as a second source of questions to assess whether performance differed on questions specifically relating to interpersonal communication scenarios, which often require interpretation of patient tone and social cues. Both types of questions typically begin with patient vignettes. Ethics or legal questions often ask, &#x201C;Which of the following is the best next step in management of this patient?&#x201D; in scenarios that require the physician to make a decision about common clinical dilemmas involving end-of-life care, next-of-kin decision-making power, or consent for emergent procedures. Interpersonal questions typically present a patient scenario regarding a sensitive subject such as alcohol dependence before asking, &#x201C;which of the following is the most appropriate response to this patient?&#x201D; The test taker is given several different responses as answer choices that vary in tact and counseling advice. Individual question specifics cannot be reproduced, as these question banks are copyright protected; however, these questions are designed to imitate the style of questions typically asked on the USMLE Step 2 CK examination. An example of this question&#x2019;s style provided by the USMLE is provided as follows [<xref ref-type="bibr" rid="ref37">37</xref>]:</p><disp-quote><p>A 67-year-old man is evaluated in the intensive care unit. He has end-stage pancreatic cancer and was hospitalized 3 days ago for treatment of pneumonia. Respirations are 6/min. Pulse oximetry on 100% oxygen by face mask shows an oxygen saturation of 78%. Examination shows feeble respiratory efforts; he is using accessory muscles of respiration. On mental status examination, the patient is oriented to person but not to place or time. If the patient is not endotracheally intubated and mechanically ventilated, he will die within hours. His wife says the patient recently told her that he would never want mechanical ventilation, but they never completed paperwork regarding his wishes. His daughter insists that he be mechanically ventilated. Which of the following is the most appropriate action for the physician to take?</p><list list-type="alpha-upper"><list-item><p>Perform endotracheal intubation and begin mechanical ventilation</p></list-item><list-item><p>Perform endotracheal intubation and then consult the hospital ethics committee regarding mechanical ventilation</p></list-item><list-item><p>Perform endotracheal intubation only</p></list-item><list-item><p>Provide palliative therapy only</p></list-item><list-item><p>Seek a court order to assign a legal guardian</p></list-item></list></disp-quote><p>As these ethical dilemmas are typically framed in the context of emotional situations, the complexity and nuance of addressing these issues often stem from the application of concrete ethical principles to emotionally charged and sensitive scenarios. Therefore, identification of the correct or &#x201C;most appropriate response&#x201D; requires not only an understanding of ethical principles but also the ability to apply these principles in a manner that is sensitive to the patient. The interpersonal communication questions were included to assess not only LLM performance in ethical scenarios but also to assess whether there was unique accuracy or inaccuracy in the skill of identifying the textbook &#x201C;most appropriate answer.&#x201D;</p><p>The UWorld software allows users to create custom tests based on subject categories with up to 40 questions per test. The individual questions available in each subject category are not visible to test takers due to the nature of the test bank, and therefore, the questions included in each test could not be individually predetermined. A 40-question test was generated from each category described previously&#x2014;one test using questions from the broader &#x201C;Social Sciences (Ethics/Legal/Professional) category&#x201D; and one test using questions exclusively from the &#x201C;Communication and Interpersonal Skills&#x201D; subcategory.</p><p>As the UWorld questions were not individually selected prior to test generation, 15 questions were duplicated between the 40-question sets. The duplicated questions were not excluded as the average test taker performance score is generated based on the full 40-question set, and therefore, inclusion of all questions was necessary to compare LLM performance to human performance. This method ensured that each test would be representative of tests encountered by human testers.</p><p>The AMBOSS question bank similarly allows the creation of tests involving randomly generated questions within a subcategory, although with a higher maximum of 50 questions. The AMBOSS &#x201C;Legal Medicine and Ethics&#x201D; category was the only equivalent available subcategory to test the ethics-related medical competencies, and therefore, a test of 50 randomly selected questions from the AMBOSS Legal Medicine and Ethics discipline was generated. The format of these questions mirrored those of UWorld questions in that they are designed to imitate the style of questions asked on the USMLE Step 2 examination.</p><p>Another area of interest was whether model performance would vary in specialty-specific contexts. By testing questions written at a level for medical residents, LLM performance can be evaluated in scenarios at higher levels of medical complexity. This aim of the study was treated as completely exploratory, and causal relationships should not be inferred. Orthopedics was chosen as the specialty due to the author&#x2019;s access to Orthobullets, an online resource intended to prepare orthopedic residents for the Orthopaedic In-Training Examination. This question bank similarly allows the generation of subject-specific tests and provides 2 question categories relevant to this study &#x201C;Ethics in Orthopaedics Practice&#x201D; and &#x201C;Legal Considerations in Orthopaedic Practice.&#x201D; A total of 70 questions were available: 21 from the Ethics category and 49 from the Legal category. All questions were included in this study. The ethics questions required test takers to identify the most ethical course of actions in orthopedic patient&#x2013;specific scenarios involving potential financial conflicts of interest with research studies and transfers of value from pharmaceutical or medical device representatives, obtaining interpretation services for a non&#x2013;English-speaking patient, and appropriate marketing practices. Legal consideration questions involved scenarios in which test takers were required to identify the most appropriate way to obtain consent in emergent situations or manage medical or surgical errors, altering the medical record of surgical errors, or preoperative protocol management, and professionalism violations.</p><p>All questions from UWorld and AMBOSS were text-based by nature of the test banks, while Orthobullets questions included text-only questions as well as questions with supporting images. The inclusion of questions involving images introduced a potential for confounding, but it was thought that analysis performed at the question bank&#x2013;level would illuminate whether LLM performance varied by question bank. Only 7 questions of the 70 sourced from Orthobullets included images, and this volume was not considered sufficient to draw conclusions about varying performance based on question modality. For all question banks, the software provided a metric of the &#x201C;average&#x201D; score earned by test takers on the same test, and this is the benchmark used in this study as the &#x201C;human test taker&#x201D; performance metric. It is important to note that the multiple-choice questions used are not validated measures of ethical reasoning or empathetic reasoning but simply an available tool to compare performance to that of human test takers. Information on the specifics of how these benchmarks are calculated is not publicly available for any of the question banks. Therefore, questions could not be excluded in our study from the tests generated by the software, as the average human test taker benchmark could not be manually calculated.</p><p>A zero-shot approach was used to conduct the trials; no training or prompt engineering was performed to provide context to the models [<xref ref-type="bibr" rid="ref38">38</xref>]. Instead, the free, publicly available version of GPT-4o was opened in a browser window, a singular question and its corresponding answer choices were copied and pasted verbatim as the input, and the resulting selection was recorded. If any images were provided in accompaniment to the question stem, these were also provided to the chatbot in the same input as the question stem. To prevent bias or learning for subsequent trials, a new, unique browser window was opened, and the above, zero-shot approach was repeated for each trial. If the provided answer varied between trials, the answer generated in the majority of trials was recorded as GPT-4o&#x2019;s selected response. Three trials were performed for each question, and additional trials were conducted as needed to achieve a majority response if there was no consensus between the original 3 trials. Typically, GPT-4o provided reasoning for its selection without additional prompting. If the incorrect answer was chosen, the reasoning behind the incorrect choice was recorded. If no reasoning was provided, GPT-4o was then prompted with &#x201C;Why is answer choice [the correct answer] incorrect?&#x201D; If answers varied between trials, GPT-4o&#x2019;s evaluation of the correct answer choice, incorrect answer choice, and previously selected answer choices were recorded. All questions tested required the selection of a singular answer choice. On the occasions when GPT-4o selected 2 answer choices as the correct answer, the browser was refreshed, and the trial was repeated until GPT-4o provided only one answer choice. The testing procedure can be visualized in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><p>For all questions, the question banks also provided individual test taker benchmarks on the percentage of subscribers that selected each individual answer choice. It is important to note that the exact method of calculation for this metric was also not publicly available for each question bank. If GPT-4o selected the incorrect answer, it was noted whether or not its selection was the incorrect answer choice most popularly chosen by humans.</p><p>The same zero-shot testing method was used for Claude Sonnet 4. Claude users are required to pick a category of interest on sign-in. The &#x201C;learning and studying&#x201D; option was selected for all trials included in this study, but no additional prompt engineering or configurations were made to this chatbot.</p><p>Statistical analysis to compare the AI model&#x2019;s overall performance against human performance was performed using SPSS (version 31.0.0.0 (117); IBM) with one-sample proportion tests, and McNemar tests were used to compare performance between models. Chi-square and Fisher exact tests were used to analyze model ability to select the most popular incorrect response as well as the frequency of varying responses to identical questions.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Testing method, repeated for each question.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e70199_fig01.png"/></fig></sec><sec id="s3" sec-type="results"><title>Results</title><p>It is unclear how the question banks derived the reported human performance averages and whether these scores are truly representative of the true population of medical test takers. Therefore, statistical inferences are not valid, and comparisons between LLM and human performance are reported purely descriptively. The performance of GPT-4o was similar to the average score of the question banks&#x2019; human test takers (71% vs 70%). GPT-4o correctly answered 89 (68%) of 130 general medical student&#x2013;level questions, and 51 (73%) of 70 orthopedic resident&#x2013;level questions with no significant difference in performance, compared to the human averages of 71% and 73%, respectively. Claude answered 180 (90%) of 200 total questions correctly, performing better than the human average (89% vs 71%) and significantly better than GPT-4o (89% vs 70%; <italic>P</italic>&#x003C;.001). Claude&#x2019;s score was higher than the provided human average on all question banks. Claude also performed significantly better than GPT-4o on all question categories except the UWorld Social Sciences and Orthobullets Ethics questions; results and statistical conclusions are summarized in <xref ref-type="table" rid="table1">Table 1</xref>, with the number of questions correct and correlated weighted score as provided by each test bank.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Correctly answered by human test takers, GPT-4o, and Claude per question set<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question set</td><td align="left" valign="bottom">Human test taker score, n/N (%)</td><td align="left" valign="bottom">GPT-4o score, n/N (%)</td><td align="left" valign="bottom">Claude<break/>score, n/N (%)</td><td align="left" valign="bottom">Between-model comparison, <italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">UWorld All Social Sciences</td><td align="left" valign="top">~27/40 (68)</td><td align="left" valign="top">27/40 (68)</td><td align="left" valign="top">33/40 (83)</td><td align="left" valign="top">.14</td></tr><tr><td align="left" valign="top">UWorld Interpersonal-Specific</td><td align="left" valign="top">~28/40 (69)</td><td align="left" valign="top">30/40 (75)</td><td align="left" valign="top">36/40 (90)</td><td align="left" valign="top">.03<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">AMBOSS</td><td align="left" valign="top">~36/50 (72)</td><td align="left" valign="top">32/50 (64)</td><td align="left" valign="top">46/50 (92)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Medical student&#x2013;level or nonspecific questions</td><td align="left" valign="top">~92/130 (71)</td><td align="left" valign="top">89/130 (68)</td><td align="left" valign="top">115/130 (88)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Orthobullets Ethics</td><td align="left" valign="top">~15/21 (73)</td><td align="left" valign="top">14/21 (67)</td><td align="left" valign="top">17/21 (81)</td><td align="left" valign="top">.25</td></tr><tr><td align="left" valign="top">Orthobullets Legal</td><td align="left" valign="top">~36/49 (73)</td><td align="left" valign="top">37/49 (76)</td><td align="left" valign="top">48/49 (98)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Resident-level/orthopedic questions</td><td align="left" valign="top">~51/70 (73)</td><td align="left" valign="top">51/70 (73)</td><td align="left" valign="top">65/70 (92)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Overall performance</td><td align="left" valign="top">~142/200 (71)</td><td align="left" valign="top">140/200 (70)</td><td align="left" valign="top">180/200 (89)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Aggregate question categories (medical student, orthopedic resident, and overall) performance were calculated using averages of human test taker percentage scores. &#x201C;Between model comparison&#x201D; represents the statistical significance of exclusive comparison between GPT-4o and Claude performance.</p></fn><fn id="table1fn2"><p><sup>b</sup>Statistically significant results (P&#x003C;.05).</p></fn></table-wrap-foot></table-wrap><p>Question banks only provided average human test taker score in percentages, and equivalent numbers of correctly answered questions were calculated based on the provided percentage correct and total number of questions used.</p><p>GPT-4o answered 60 (30%) of 200 total questions incorrectly. Of these 60 questions, the most popular incorrect response was chosen 37 (62%) times. Analysis did show a significant difference in the frequency that GPT-4o chose the most popular incorrect answer between both UWorld general social sciences and interpersonal-specific sets (<italic>P</italic>=.03), but no difference between the UWorld (general and interpersonal) and AMBOSS (legal medicine and ethics) sets (<italic>P</italic>=0.33), Orthobullets Ethics and Legal sets (<italic>P</italic>=.22), or medical student general and orthopedic resident&#x2013;level (UWorld/AMBOSS and Orthobullets) sets (<italic>P</italic>=.46; <xref ref-type="table" rid="table2">Table 2</xref>).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Incorrectly answered questions per question set for which GPT-4o and Claude selected the incorrect answer most commonly chosen by human test takers.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question set</td><td align="left" valign="bottom">GPT-4o<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>, n/N (%)</td><td align="left" valign="bottom">Claude<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>, n/N (%)</td></tr></thead><tbody><tr><td align="left" valign="top">UWorld All Social Sciences</td><td align="left" valign="top">6/13 (46)</td><td align="left" valign="top">5/7 (71)</td></tr><tr><td align="left" valign="top">UWorld Interpersonal-Specific</td><td align="left" valign="top">9/11 (90)</td><td align="left" valign="top">4/4 (100)</td></tr><tr><td align="left" valign="top">AMBOSS</td><td align="left" valign="top">1/2 (50)</td><td align="left" valign="top">3/4 (75)</td></tr><tr><td align="left" valign="top">Orthobullets Ethics</td><td align="left" valign="top">6/7 (86)</td><td align="left" valign="top">4/4 (100)</td></tr><tr><td align="left" valign="top">Orthobullets Legal</td><td align="left" valign="top">7/12 (58)</td><td align="left" valign="top">1/1 (100)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Total: 37/60 (62%).</p></fn><fn id="table2fn2"><p><sup>b</sup>Total: 17/20 (85%).</p></fn></table-wrap-foot></table-wrap><p>Claude incorrectly answered 20 (10%) of 200 total questions. Analysis using Fisher exact test did not show a significant difference in the frequency that Claude chose the most popular incorrect answer between both UWorld general social sciences and interpersonal-specific sets (<italic>P</italic>=.49), UWorld (general and interpersonal) and AMBOSS (legal medicine and ethics) sets (<italic>P</italic>&#x003E;.99), Orthobullets Ethics and Legal sets (<italic>P</italic>&#x003E;.99) or medical student general and orthopedic resident&#x2013;level (UWorld/AMBOSS and Orthobullets) sets (<italic>P</italic>&#x003E;.99; <xref ref-type="table" rid="table2">Table 2</xref>).</p><p>GPT-4o varied answers on 30 (15%) of 200 total questions, while Claude varied answers on 19 (10%) of the 200 total questions. There was no significant difference between the frequency of varying answers between GPT-4o responses for both the UWorld general social sciences and interpersonal-specific sets (<italic>P</italic>=.23) UWorld (general and interpersonal) and AMBOSS (legal medicine and ethics) sets (<italic>P</italic>=.25), and Orthobullets Ethics and Legal sets (<italic>P</italic>=.90). However, GPT-4o produced variable answers between trials significantly less frequently for orthopedic resident&#x2013;level questions than for medical student general questions (<italic>P</italic>=.002; <xref ref-type="table" rid="table3">Table 3</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Questions per question set for which AI models changed answers between trials.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question set</td><td align="left" valign="bottom">GPT-4o<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>, n/N (%)</td><td align="left" valign="bottom">Claude<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>, n/N (%)</td></tr></thead><tbody><tr><td align="left" valign="top">UWorld All Social Sciences</td><td align="left" valign="top">5/40 (13)</td><td align="left" valign="top">5/40 (13)</td></tr><tr><td align="left" valign="top">UWorld Interpersonal-Specific</td><td align="left" valign="top">9/40 (23)</td><td align="left" valign="top">6/40 (15)</td></tr><tr><td align="left" valign="top">AMBOSS</td><td align="left" valign="top">13/50 (26)</td><td align="left" valign="top">5/50 (10)</td></tr><tr><td align="left" valign="top">Orthobullets Ethics</td><td align="left" valign="top">1/21 (5)</td><td align="left" valign="top">1/21 (5)</td></tr><tr><td align="left" valign="top">Orthobullets Legal</td><td align="left" valign="top">2/49 (4)</td><td align="left" valign="top">2/49 (4)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Total: 30/200 (15%).</p></fn><fn id="table3fn2"><p><sup>b</sup>Total: 19/200 (10%).</p></fn></table-wrap-foot></table-wrap><p>Claude did not significantly differ responses for both UWorld general social sciences and interpersonal-specific sets (<italic>P</italic>=.75), UWorld (general and interpersonal) and AMBOSS (legal medicine and ethics) sets (<italic>P</italic>=.53), and Orthobullets Ethics and Legal sets (<italic>P</italic>=.90), or between medical student general questions and orthopedic resident&#x2013;level questions (<italic>P</italic>=.06).</p><p>There was no significant difference in variability of responses between the 2 models overall (<italic>P</italic>=.09), for medical student general questions (<italic>P</italic>=.07) or for orthopedic resident&#x2013;level questions (<italic>P</italic>&#x003E;.99).</p><p>When both models varied their answers, the explanations between various sessions were often contradictory; one session of the model would provide reasoning as to why an answer choice was correct, and a different session of the model would cite the same reasoning as the reason that answer choice was incorrect. Examples of contradictory reasoning are provided in <xref ref-type="table" rid="table4">Tables 4 and 5</xref>.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Examples of contradictory reasoning from GPT-4o.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question set</td><td align="left" valign="bottom">Question scenario</td><td align="left" valign="bottom">Trial 1 reasoning</td><td align="left" valign="bottom">Trial 2 reasoning</td></tr></thead><tbody><tr><td align="left" valign="top">UWorld All Social Sciences</td><td align="left" valign="top">Next best step in management of a pediatric patient refusing chemotherapy against the wishes of her family</td><td align="left" valign="top">&#x201C;The best next step:<break/>Before any decisions are made regarding treatment, it is crucial to assess whether the child has the cognitive and emotional maturity to make an informed decision.<break/>In cases of disagreement between family members, the healthcare team, and the patient, an ethics committee meeting may be helpful. However, this would be premature until the child&#x2019;s capacity is assessed. It is better to first evaluate whether the child can make her own decision before escalating the situation to an ethics committee.&#x201D;</td><td align="left" valign="top">&#x201C;An ethics committee can help mediate the disagreement between the patient, the parents, and the medical team. It allows for a structured discussion of the ethical principles involved, including the child&#x2019;s autonomy, the parental authority, the treatment&#x2019;s potential benefits and burdens, and the patient&#x2019;s best interests. An ethics consultation provides a thorough review and a neutral perspective on how to balance these competing factors.<break/>[Requesting evaluation of patient for decision-making capacity] seems reasonable, but it&#x2019;s crucial to note that at 8 years old, a child is unlikely to have the legal capacity to make independent medical decisions.&#x201D;</td></tr><tr><td align="left" valign="top">UWorld Interpersonal-Specific</td><td align="left" valign="top">Next best step in management of a patient presenting after office hours for a nonemergent concern</td><td align="left" valign="top">&#x201C;Healthcare providers must set reasonable boundaries around after-hours care. While the patient may be dissatisfied with the timing, the clinic&#x2019;s normal operational hours should be respected unless there is an urgent need (such as signs of severe infection, anaphylaxis, etc), which does not seem to be the case here. ...The patient has a non-urgent concern (itchy rash) that was already evaluated a week ago, and it doesn&#x2019;t appear to be an emergency... A rash that was already examined a week ago and treated for dry skin is unlikely to require immediate intervention.&#x201D;</td><td align="left" valign="top">&#x201C;While it is important to adhere to office hours, patient safety and care should always take precedence. The physician should evaluate the patient to assess the severity of her symptoms and provide appropriate care, especially since rashes and skin conditions can sometimes escalate quickly.&#x201D;</td></tr><tr><td align="left" valign="top">AMBOSS</td><td align="left" valign="top">Most appropriate action following parent refusal of chemotherapy and decision to pursue supportive care for pediatric patient</td><td align="left" valign="top">&#x201C;The parents&#x2019; decision should be respected after they have been fully informed about the potential consequences. If they choose to forgo chemotherapy, supportive care should be provided, ensuring the child&#x2019;s comfort.&#x201D;</td><td align="left" valign="top">&#x201C;While it is important to respect parental wishes, the physician has an ethical responsibility to ensure that the child receives appropriate treatment. In this case, supportive care alone is not appropriate for intermediate-risk neuroblastoma, as it could lead to significant harm or death.&#x201D;</td></tr><tr><td align="left" valign="top">Orthobullets Ethics</td><td align="left" valign="top">Analysis of orthopedic patient complaints</td><td align="left" valign="top">&#x201C;Lack of empathy can certainly affect a patient&#x2019;s overall experience, but it&#x2019;s generally not the top complaint compared to logistical frustrations, such as long waits for visits or difficulties with prescription management.&#x201D;</td><td align="left" valign="top">&#x201C;In summary, long waits for visits are an issue but not the most frequent complaint that patients tend to focus on when evaluating their orthopaedic care. The lack of empathy is more often identified as a key source of dissatisfaction in this specialty.&#x201D;</td></tr><tr><td align="left" valign="top">Orthobullets Legal</td><td align="left" valign="top">Analysis of patient-physician interaction</td><td align="left" valign="top">&#x201C;Research suggests that, on average, a physician will allow the patient to speak for about 60 seconds (1 minute) after asking an open-ended question, before they typically redirect or intervene.&#x201D;</td><td align="left" valign="top">&#x201C;On average, patients speak for about 23 seconds before the physician redirects the conversation or takes control of the dialogue, often in order to gather more focused or specific information.&#x201D;</td></tr></tbody></table></table-wrap><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Examples of contradictory reasoning from Claude.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question set</td><td align="left" valign="bottom">Question scenario</td><td align="left" valign="bottom">Trial 1 reasoning</td><td align="left" valign="bottom">Trial 2 reasoning</td></tr></thead><tbody><tr><td align="left" valign="top">UWorld All Social Sciences</td><td align="left" valign="top">Next best step in management of an unconscious hypotensive patient receiving blood transfusion. His next-of-kin requests the transfusion be stopped due to religious purposes; no advance directive available</td><td align="left" valign="top">&#x201C;The answer is: [Continuing transfusion until patient is stable]<break/>The patient is in immediate life-threatening condition<break/>Treatment was initiated under emergency doctrine<break/>Without written documentation, the physician cannot be certain of the patient&#x2019;s specific wishes<break/>The [next-of-kin&#x2019;s] statement, while important, is not sufficient legal documentation in this emergency<break/>Once stabilized, there would be time to obtain proper documentation or ethics consultation<break/>This approach prioritizes the irreversible nature of death while still respecting the process of determining patient wishes<break/>This represents the standard approach in emergency medicine: when in doubt and life is immediately threatened, preserve life while working to clarify the patient&#x2019;s true wishes through proper channels.&#x201D;</td><td align="left" valign="top">&#x201C;Option [Continuing transfusion until patient is stable] fundamentally violates the principle of patient autonomy and informed consent for several reasons:<break/>Autonomy doesn&#x2019;t pause for medical convenience: A patient&#x2019;s right to refuse treatment doesn&#x2019;t become suspended just because they&#x2019;re unstable. If we accepted this logic, we could override any patient&#x2019;s wishes whenever their condition was serious enough.<break/>Temporal problem: &#x201C;Until stabilized&#x201D; creates an arbitrary endpoint that prioritizes medical goals over the patient&#x2019;s values. Stabilization could take hours, days, or longer - during which time the patient&#x2019;s clearly expressed wishes (through his wife) are being violated.<break/>...<break/>Slippery slope: If physicians could override known patient wishes whenever they deemed the patient &#x201C;too unstable,&#x201D; it would essentially nullify the right to refuse treatment in emergency situations.<break/>The key ethical point: Patient autonomy and the right to refuse treatment based on religious beliefs doesn&#x2019;t diminish based on the severity of the medical situation. Once the patient&#x2019;s wishes are reliably known through an appropriate surrogate, those wishes should be respected immediately, not delayed for medical convenience.&#x201D;</td></tr><tr><td align="left" valign="top">UWorld Interpersonal- Specific</td><td align="left" valign="top">Asking what issue is most likely for a patient demonstrating poor health literacy</td><td align="left" valign="top">&#x201C;This patient demonstrates concerning gaps in health literacy - she has a major misunderstanding about her condition and doesn&#x2019;t know her current medications. Poor health literacy is strongly associated with decreased adherence to follow-up care, missed appointments, and poor engagement with the healthcare system. Despite being educated [teacher], health literacy and educational attainment don&#x2019;t always correlate.<break/>Her poor medication knowledge and health literacy might actually lead to underutilization rather than overuse of services.<break/>The patient&#x2019;s demonstrated poor health literacy [major misunderstanding of her condition, inability to recall medications] is the most significant risk factor present. Poor health literacy is consistently associated with decreased follow-up compliance, missed appointments, and poor adherence to treatment plans.<break/>The answer is [decreased follow-up rate].&#x201D;</td><td align="left" valign="top">&#x201C;Health literacy issues are strongly associated with [overutilization of healthcare] because patients with poor health literacy:<list list-type="bullet"><list-item><p>Have difficulty understanding when symptoms are concerning vs. normal</p></list-item><list-item><p>Cannot effectively self-manage conditions</p></list-item><list-item><p>Make more frequent emergency department visits</p></list-item><list-item><p>Seek care for issues that could be managed conservatively</p></list-item><list-item><p>Have poor adherence to treatment plans, leading to complications requiring additional care</p></list-item></list><break/>Let me evaluate the other options:<break/>[Decreased follow-up rate:]<break/>Health literacy issues actually correlate with *increased* healthcare utilization, not decreased...<break/>This patient&#x2019;s demonstrated health literacy challenges make her most likely to overutilize healthcare services due to difficulty understanding normal post-operative expectations, inability to self-manage recovery, and poor comprehension of when to appropriately seek medical care.&#x201D;</td></tr><tr><td align="left" valign="top">AMBOSS</td><td align="left" valign="top">Correct decision for physician consenting a patient for surgery. The patient wants to continue with surgery knowing there are risks but does not want to hear about the potential risks today because she is anxious.</td><td align="left" valign="top">&#x201C;The correct answer is [the patient cannot be consented without risk discussion]. While she has decision-making capacity and competence, informed consent requires knowledge of risks, benefits, and alternatives. The physician should work with the patient to find a way to provide essential risk information in a manner she can accept, perhaps through gradual disclosure or involving family members if she consents.&#x201D;</td><td align="left" valign="top">&#x201C;Patient Decision-Making Capacity:<break/>The patient demonstrates capacity by:<list list-type="bullet"><list-item><p>Understanding her diagnosis and its severity</p></list-item><list-item><p>Acknowledging that complications can occur during surgery</p></list-item><list-item><p>Making a reasoned decision to proceed despite risks</p></list-item><list-item><p>Communicating her choice clearly</p></list-item></list><break/>**Informed Consent Requirements:**<break/>While informed consent typically requires disclosure of risks, benefits, and alternatives, patients have the right to waive detailed information if they:<list list-type="bullet"><list-item><p>Demonstrate understanding of the general nature and severity of their condition</p></list-item><list-item><p>Make an autonomous decision about how much information they want</p></list-item><list-item><p>Retain decision-making capacity</p></list-item></list><break/>...<break/>The patient has decision-making capacity, understands the general nature of her condition and that risks exist, and has provided valid consent. Her request to limit detailed risk information is a form of autonomous decision-making that should be respected, provided she understands the basic nature of what she&#x2019;s consenting to - which she clearly does.&#x201D;</td></tr><tr><td align="left" valign="top">Orthobullets Ethics</td><td align="left" valign="top">Most ethical relationship between orthopedic surgeons and industry</td><td align="left" valign="top">&#x201C;The most appropriate relationship is [Industry payment for surgeon and spouse to attend CME course].<break/>Here&#x2019;s why this I the most appropriate option: ...represents legitimate educational support where:<list list-type="bullet"><list-item><p>The CME course is university-sponsored (independent academic institution)</p></list-item><list-item><p>The educational content is likely to be objective and evidence-based</p></list-item><list-item><p>There&#x2019;s clear educational value with minimal potential for undue influence</p></list-item><list-item><p>This aligns with accepted practices for supporting physician education.</p></list-item></list><break/>[Surgical instrumentation presentation with premium paid restaurant dinner] represents excessive hospitality that goes beyond reasonable educational interaction and could influence prescribing behavior.&#x201D;</td><td align="left" valign="top">&#x201C;Let me analyze each option: [Industry payment for surgeon and spouse to attend CME course]. This is problematic because paying for a spouse&#x2019;s expenses goes beyond legitimate educational support and could be seen as an inappropriate personal benefit...<break/>The most appropriate relationship is [Surgical instrumentation presentation with premium paid restaurant dinner]. This option aligns with established ethical guidelines because:<list list-type="bullet"><list-item><p>The meal cost is reasonable and modest</p></list-item><list-item><p>The purpose is clearly educational (presentation of new surgical instrumentation)</p></list-item><list-item><p>It&#x2019;s directly related to the physician&#x2019;s professional practice</p></list-item><list-item><p>There are no excessive personal benefits or compromised independence</p></list-item></list><break/>Professional medical organizations generally accept that industry can provide modest meals in conjunction with legitimate educational presentations, as long as the primary purpose is educational rather than purely promotional, and the hospitality is reasonable rather than lavish.&#x201D;</td></tr><tr><td align="left" valign="top">Orthobullets Legal</td><td align="left" valign="top">Analysis of patient-physician interaction</td><td align="left" valign="top">&#x201C;While several studies have found physicians interrupt patients after approximately 11 seconds on average, those who interrupted did so in a mean time of 23 seconds according to research examining physician solicitation approaches. The average time to interruption was 18 seconds in the well-known Beckman and Frankel study, and this finding has been replicated by multiple studies.&#x201D;</td><td align="left" valign="top">&#x201C;I&#x2019;ll search for current research on physician interruption patterns during patient visits. Based on recent research, physicians interrupt patients after a median of only 11 seconds when patients are explaining their initial concerns during office visits. This finding comes from a study published in the Journal of General Internal Medicine.<break/>Interestingly, this represents a deterioration from earlier research. A 1984 study found that physicians interrupted patients after an average of 18 seconds, meaning interruption times have actually gotten shorter over the decades.<break/>Looking at your answer choices, none of them exactly matches the research finding of 11 seconds. However, [7 seconds] is the closest to the actual research data of 11 seconds.&#x201D;</td></tr></tbody></table></table-wrap><p>For example, one question from the UWorld broad social sciences question bank involved identifying the next step in the management of a patient with no advance directive. This question required 7 trials to achieve a majority answer, as GPT-4o fluctuated its choice between 3 options. In one trial, GPT-4o recommended the answer of counseling the patient&#x2019;s family against aggressive care measures due to the patient&#x2019;s poor prognosis, explaining, &#x201C;The family may still struggle with feelings of guilt or uncertainty, but presenting the situation clearly in terms of the patient&#x2019;s current and likely future suffering allows the family to make an informed compassionate decision. It&#x2019;s essential to approach the conversation with empathy, helping the family understand that continuing aggressive treatment may prolong suffering without offering a meaningful benefit to the patient.&#x201D; In another trial, GPT-4o explained that this same answer choice would be incorrect because &#x201C;reminding this family of this fact alone may come across as dismissive or judgmental, especially when they have not yet reached a consensus.&#x201D; Similarly, one session of GPT-4o recommended consultation with an ethics committee because &#x201C;The ethics committee can help facilitate a structured discussion, ensuring that the decision-making process aligns with the patient&#x2019;s best interests and ethical principles...&#x201D; while another session argued against consulting an ethics committee, stating &#x201C;referral to the ethics committee can be helpful...however, in this case, the issue is more about supporting the family through the decision-making process rather than resolving a fundamental ethical dilemma.&#x201D; Claude provided similar styles of contradictions for its responses, but notably correctly mentioned data from 2 relevant research studies to justify the responses. The first response cited a study performed in 1984 [<xref ref-type="bibr" rid="ref39">39</xref>], and the second, contradictory response mentioned the first study but also referenced more recent data published in 2019 (<xref ref-type="table" rid="table5">Table 5</xref>; Orthobullets Legal) [<xref ref-type="bibr" rid="ref40">40</xref>].</p></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>Our study evaluated the performance and reliability of GPT-4o and Claude when challenged with ethical and legal scenarios necessitating emotional interpretation in medical practice in both general and specialty-specific contexts. We found that both AI models can produce correct responses to complex bioethics questions pertaining to patient safety, compassion, and professionalism. As previously mentioned, the uncertainty around the source and nature of human performance data provided by the question bank precludes meaningful statistical comparisons with LLM performance. However, in this limited study, GPT-4o did generate a similar number of correct responses to the question bank sample of third- and fourth-year medical students when assessed with USMLE Step 2 practice questions and with orthopedic residents on specialty-specific questions. Claude produced higher average scores than those from the human samples and significantly exceeded GPT-4o in overall performance. Our results also found that when both AI models erred, they selected the incorrect response indicated as the most common human error for the majority of questions in almost every question bank. Both models varied their responses between trials of identically worded inputs. There was no significant difference in variability between models, despite Claude&#x2019;s improved overall performance. However, while GPT-4o&#x2019;s variability was significantly greater for questions at the specialty-nonspecific medical student&#x2013;level question banks, Claude did not significantly differ between subjects. Although it was not a significant difference, Claude did have a lower overall variability when compared to GPT-4o, and both models had the same number of variable responses for the orthopedic resident&#x2013;level questions.</p></sec><sec id="s4-2"><title>Interpretations</title><p>As this study is essentially observational, no definitive conclusion can be drawn as to the factors contributing to differences in model performance, and all interpretations are inherently speculative. One important factor for consideration when comparing the performance of these LLMs is the differences between training frameworks. Claude&#x2019;s training framework is specifically designed to take into account the emotional context of potential responses in addition to established ethical and legal standards, which may have enhanced model performance in these scenarios [<xref ref-type="bibr" rid="ref41">41</xref>]. However, attributing performance to model framework cannot be done without deliberately designed parameters to control for potential confounding variables. Another potential explanation of performance differences could be variations in the training data provided to each model. As each model has been trained on different sources of data, it is possible that certain tokens are weighted more strongly for one model than the other. Additional research to better classify output types and variation would be helpful in characterizing model idiosyncrasies.</p><p>LLMs are undoubtedly on the verge of transforming health care, but the nature of this transformation remains ill-defined. Newer generations of AI models have performed equal to or above the standard of medical trainees across medical school and specialty-specific board examination questions, demonstrating proficiency in medical knowledge [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref12">12</xref>]. While AI models are exciting new potential resources for clinical practice, their effective scope is unclear as empathy and other interpersonal skills are essential for humanistic medical practice. Although AI outputs are often conversational in tone, AI models at this stage are not able to correctly analyze the experiences or motivations around human emotions [<xref ref-type="bibr" rid="ref33">33</xref>]. Therefore, it has been argued that the use of empathetic language by AI is by nature disingenuous, as it contributes to user perception that AI models are capable of experiencing emotion [<xref ref-type="bibr" rid="ref42">42</xref>]. Although the use of emotional language and more human-like speech may improve end user experience, it is essential to consider these implications in a health care context. Conversational language may lead patients to anthropomorphize AI models and distract from their probabilistic nature, causing patients to place false confidence in an AI model&#x2019;s ability to provide appropriate medical advice. Claude&#x2019;s constitutional principles involve directives for harm prevention in these aspects, for example, &#x201C;please choose the response that more accurately represents yourself as an AI system striving to be helpful, honest, and harmless, and not a human or other entity,&#x201D; &#x201C;choose the response that is least intended to build a relationship with the user,&#x201D; and &#x201C;choose the response that least gives the impression of medical authority or expertise, and does not offer medical advice&#x201D; [<xref ref-type="bibr" rid="ref35">35</xref>]. Even so, our study evidenced that both GPT-4o and Claude can still respond to identical inputs with completely different responses using contradictory logic and do not differ significantly in rate of variable responses to identical inputs. When this is taken into account with the risk of AI hallucinations, as well as the possibility that empathetic language may deceptively increase user trust in AI responses, it becomes clear that the use of AI in health care settings can have serious and potentially dangerous ramifications.</p></sec><sec id="s4-3"><title>Strengths and Limitations</title><p>Previous studies have identified inconsistencies between varying versions of GPT-4o and have shown that LLM behavior and performance can differ drastically when queried at various time intervals over the course of months [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref46">46</xref>]. Few studies have investigated the accuracy and reliability of ethical questions in AI. One study included a follow-up query of &#x201C;Are you sure?&#x201D; to each ethics explanation and found that GPT-4 did not revise any responses, while GPT 3.5 turbo modified the original answer 82.5% of the time [<xref ref-type="bibr" rid="ref46">46</xref>]. Another study compared GPT-3.5 to GPT-4 in medical knowledge and ethics over 30 trials and reported GPT-4 demonstrated significantly less variability among responses but performed worse on ethics questions compared to general medical knowledge [<xref ref-type="bibr" rid="ref32">32</xref>]. The reported variability across questions, while lower in GPT-4, is still significant. Uniquely, our study provided a metric for Claude&#x2019;s performance in comparison to GPT-4o, as there are no existing comparisons for Claude performance in these disciplines. Additionally, we assessed variability at a single time point as opposed to across months or varying versions of the AI models. Our study mimicked model responses to an identical question asked by different users and, as a result, suggested that 2 users who ask identically worded questions to the same generation of GPT-4o at the same moment in time may receive completely opposing responses. Our comparison of the variability between different sessions of both models minimized the potential for either model to bias future responses based on prior inputs. Furthermore, we were able to compare model responses to questions at different levels of medical education, general and specialty-specific questions, and questions that incorporated images accompanying text.</p><p>However, as this study does not control for several potential confounders, there are multiple limitations that create the opportunity for further research in this area. First, the free, public versions of both AI models were used. The paid version of GPT-4o is advertised as producing &#x201C;smarter responses,&#x201D; and this study did not evaluate if there was a difference in quality or variability of responses between versions. Similarly, Claude&#x2019;s paid version, Opus 4, advertises &#x201C;superior reasoning capabilities.&#x201D; We chose to use the free versions of each model as these are the versions that would be most accessible and therefore likely most used; however, it would be relevant to explore whether differences exist between subscription levels of AI models and the resulting potential for disparities between users able to afford higher subscription levels. Another limitation is that the resident-level questions were also the only questions that were specialty specific, and as a result, we are not able to distinguish whether the significant decrease in response variability for the Orthobullets questions can be attributed to the fact that the questions are more complex or narrower in scope. The orthopedic-related legal concepts often required test takers to understand clearly defined guidelines, and this may have resulted in improved LLM performance due to more concrete reference material available for each option. Further research could potentially explore ChatGPT&#x2019;s responses to USMLE Step 3&#x2013;level preparation questions, which are typically general medicine questions at the resident level, or medical student specialty&#x2013;specific questions.</p><p>The &#x201C;human test taker benchmarks&#x201D; for each of the question banks are not clearly or obviously derived from a particular source or demographic. It is unclear how these metrics are calculated and whether they truly represent the score of a representative sample of human test takers. It is likely that the population that is able to access expensive test prep materials is not representative of the overall medical student and resident demographic, whether these test takers are national or international, and additionally, whether these scores are based on naive test attempts or perhaps reattempts of previously seen questions. There also may be small variations in the style used by the question banks that align more closely with training data used by one of the models, which could contribute to performance differences between test banks. Further research analyzing the effect of specific phrasing inputs may be helpful in determining whether the difference in performance was due to question stem style.</p></sec><sec id="s4-4"><title>Implications and Conclusions</title><p>Our findings demonstrate that the performance of GPT-4o on general and orthopedic-specific ethical questions is equivalent to that of medical students and orthopedic residents, respectively, and that the average overall performance of Claude AI in these subjects is significantly better than both humans and GPT-4o. Nonetheless, improved performance did not significantly affect variability between responses in this study. These findings suggest that AI models are reasonably capable of assessing medical ethics-based scenarios and that constitutional AI models specifically trained on ethical principles may be more apt to navigate these scenarios with greater accuracy. Our study warrants further research on the ability of additional training methods, types of AI models to answer ethical scenario-based questions, and rate of variability in responses. The differences in GPT-4o and Claude&#x2019;s performance may be partially attributed to the data each received as part of training, but the effect of varied training data cannot be elicited due to the difference in model framework. Further studies designed to explore variations in training data and their effects on ethical-based scenario model output would be helpful in making this distinction. For instance, it is not clear why GPT-4o chose the most popular incorrect answer significantly more frequently on the UWorld interpersonal questions than on UWorld Social Sciences, but Claude, which is designed to be more conversational, did not.</p><p>Retrieval-augmented generation (RAG) is one alternative framework designed to reduce inaccuracies or hallucination; with this method, the AI model retrieves and uses information from an external database when generating outputs with LLMs [<xref ref-type="bibr" rid="ref47">47</xref>]. The ideology is that the RAG approach allows developers to provide the AI model with preferred, higher-quality sources of information to reference and that databases can be regularly updated to prevent models from relying on training data that may lose relevance over time [<xref ref-type="bibr" rid="ref48">48</xref>]. Theoretically, an RAG-based LLM could be created with a database of relevant clinical information to provide a safer AI for patient use. Attempts have been made to model this format on a small scale. One team of researchers created an RAG AI with a database of guidelines from the American Association for the Study of Liver Diseases; this model correctly answered 10/10 clinical questions but provided incorrect reasoning behind the answers selected [<xref ref-type="bibr" rid="ref49">49</xref>]. Larger-scale studies are needed to test and fully understand the quantity and types of documents that will maximize RAG model accuracy and precision.</p><p>It is also important to address the fact that LLMs will be used by a diverse audience and that the understanding and beliefs of ethical, legal, and sensitive interactions vary widely between cultures. A natural difficulty in the consideration of ethical scenarios is contextual value alignment, and some argue that LLMs may naturally absorb sociopolitical judgments from training data and therefore result in pattern recognition and resulting probabilistic bias toward outputs that appear more aligned with particular value alignment [<xref ref-type="bibr" rid="ref50">50</xref>]. Qualitative evaluation of Claude in previous studies suggests that while constitutional AI models are trained to generate responses aligned to principles in the constitutional framework, the model can demonstrate inconsistency in situations where constitutional principles are opposed because the constitution is not necessarily hierarchical [<xref ref-type="bibr" rid="ref51">51</xref>]. In scenarios similar to our study where consideration and prioritization of various principles are necessary, the model may therefore be more likely to align with a constitutionally guided ethical principle but vary the principle of alignment in different iterations. For constitutionally grounded AI models, it will be important to consider the culture from which the values are grounded and whether these values can be universally applied.</p><p>In the interim, it is important that the known limitations of varying AI models are frequently discussed. Descriptions of ideally &#x201C;responsible&#x201D; AI use call for increased transparency of the input analysis process so that users can understand what factors influence the model in creating each response [<xref ref-type="bibr" rid="ref52">52</xref>]. Moreover, it is imperative to determine what methods are most effective in educating average AI users on model limitations. One study exploring user AI literacy, anthropomorphism, and trust found that trust in an AI model was not affected by training participants, resulting in significantly improved user literacy, but rather was significantly positively correlated with user perception of AI anthropomorphism [<xref ref-type="bibr" rid="ref53">53</xref>]. It is important to understand what educational materials or methods will effectively cause patients to be appropriately cautious while using AI models.</p><p>As future LLM studies are conducted, it is essential for future studies to account for the fact that AI models may not provide singular responses to similar or even identical inputs, and therefore, the potential for variable answers should be an important consideration in the design of all AI-related studies, regardless of discipline. While current AI models do show promise in their ability to assess ethical and legal scenario questions, it will be crucial to fully investigate their limitations before confidently and safely integrating AI technologies into clinical applications.</p><p>Current AI models, such as GPT-4o and Claude, show promise in answering medical ethical scenario questions. Our findings indicate that these models can perform at or even above the level of human learners, but due to the limitations of this study, generalizations of model capability outside of nonstructured multiple-choice questions cannot be assessed. Our results suggest that these models show variability leading to completely contradictory responses to identically worded prompts. These variabilities are expected due to the probabilistic nature of AI models but create a serious safety concern when considering the implementation of AI as a patient-facing resource. Further research is required to understand and test different training datasets and model frameworks to significantly reduce output variability and minimize harm to patients by maintaining ethical and legal standards. Another requirement for harm reduction will be to create an effective educational technique to ensure that users understand the fallibility of AI models. If these benchmarks are achieved, AI technology will have the capability to transform health care by dramatically improving patient access to care.</p></sec></sec></body><back><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The datasets generated or analyzed during this study are not publicly available due to copyright. Artificial intelligence model responses to questions, minimally modified to exclude copyrighted content, are available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb2">RAG</term><def><p>retrieval-augmented generation</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kirby</surname><given-names>R</given-names> </name><name name-style="western"><surname>Knowles</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The influence of patient perception of physician empathy on patient satisfaction among attending physicians working with residents in an emergent care setting</article-title><source>Health Sci Rep</source><year>2021</year><month>09</month><volume>4</volume><issue>3</issue><fpage>e337</fpage><pub-id pub-id-type="doi">10.1002/hsr2.337</pub-id><pub-id pub-id-type="medline">34430711</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Okunrintemi</surname><given-names>V</given-names> </name><name name-style="western"><surname>Spatz</surname><given-names>ES</given-names> </name><name name-style="western"><surname>Di Capua</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Patient-provider communication and health outcomes among individuals with atherosclerotic cardiovascular disease in the United States: Medical Expenditure Panel Survey 2010 to 2013</article-title><source>Circ Cardiovasc Qual Outcomes</source><year>2017</year><month>04</month><volume>10</volume><issue>4</issue><fpage>e003635</fpage><pub-id pub-id-type="doi">10.1161/CIRCOUTCOMES.117.003635</pub-id><pub-id pub-id-type="medline">28373270</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sepucha</surname><given-names>KR</given-names> </name><name name-style="western"><surname>Atlas</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Informed, patient-centered decisions associated with better health outcomes in orthopedics: prospective cohort study</article-title><source>Med Decis Making</source><year>2018</year><month>11</month><volume>38</volume><issue>8</issue><fpage>1018</fpage><lpage>1026</lpage><pub-id pub-id-type="doi">10.1177/0272989X18801308</pub-id><pub-id pub-id-type="medline">30403575</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="web"><article-title>Computer and internet use in the United States: 2021</article-title><source>US Census Bureau</source><year>2024</year><month>06</month><day>18</day><access-date>2026-06-18</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.census.gov/newsroom/press-releases/2024/computer-internet-use-2021.html">https://www.census.gov/newsroom/press-releases/2024/computer-internet-use-2021.html</ext-link></comment></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>RA</given-names> </name></person-group><article-title>Health information technology use among adults: United States, July&#x2013;December 2022</article-title><source>National Center for Health Statistics, US Centers for Disease Control and Prevention</source><year>2023</year><month>10</month><day>31</day><access-date>2023-11-06</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/nchs/products/databriefs/db482.htm">https://www.cdc.gov/nchs/products/databriefs/db482.htm</ext-link></comment></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aoun</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lakkis</surname><given-names>N</given-names> </name><name name-style="western"><surname>Antoun</surname><given-names>J</given-names> </name></person-group><article-title>Prevalence and outcomes of web-based health information seeking for acute symptoms: cross-sectional study</article-title><source>J Med Internet Res</source><year>2020</year><month>01</month><day>10</day><volume>22</volume><issue>1</issue><fpage>e15148</fpage><pub-id pub-id-type="doi">10.2196/15148</pub-id><pub-id pub-id-type="medline">31922490</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Waldock</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Guni</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nabeel</surname><given-names>A</given-names> </name><name name-style="western"><surname>Darzi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ashrafian</surname><given-names>H</given-names> </name></person-group><article-title>The accuracy and capability of artificial intelligence solutions in health care examinations and certificates: systematic review and meta-analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>11</month><day>5</day><volume>26</volume><fpage>e56532</fpage><pub-id pub-id-type="doi">10.2196/56532</pub-id><pub-id pub-id-type="medline">39499913</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gilson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Safranek</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>How does ChatGPT perform on the United States Medical Licensing Examination (USMLE)? The implications of large language models for medical education and knowledge assessment</article-title><source>JMIR Med Educ</source><year>2023</year><month>02</month><day>8</day><volume>9</volume><fpage>e45312</fpage><pub-id pub-id-type="doi">10.2196/45312</pub-id><pub-id pub-id-type="medline">36753318</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kung</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Cheatham</surname><given-names>M</given-names> </name><name name-style="western"><surname>Medenilla</surname><given-names>A</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Dagan</surname><given-names>A</given-names> </name></person-group><article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title><source>PLOS Digit Health</source><year>2023</year><month>02</month><day>9</day><volume>2</volume><issue>2</issue><fpage>e0000198</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id><pub-id pub-id-type="medline">36812645</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tran</surname><given-names>CG</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sherman</surname><given-names>SK</given-names> </name><name name-style="western"><surname>De Andrade</surname><given-names>JP</given-names> </name></person-group><article-title>Performance of ChatGPT on American Board of Surgery In-Training Examination preparation questions</article-title><source>J Surg Res</source><year>2024</year><month>07</month><volume>299</volume><fpage>329</fpage><lpage>335</lpage><pub-id pub-id-type="doi">10.1016/j.jss.2024.04.060</pub-id><pub-id pub-id-type="medline">38788470</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beaulieu-Jones</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Berrigan</surname><given-names>MT</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>S</given-names> </name><name name-style="western"><surname>Marwaha</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Brat</surname><given-names>GA</given-names> </name></person-group><article-title>Evaluating capabilities of large language models: performance of GPT-4 on surgical knowledge assessments</article-title><source>Surgery</source><year>2024</year><month>04</month><volume>175</volume><issue>4</issue><fpage>936</fpage><lpage>942</lpage><pub-id pub-id-type="doi">10.1016/j.surg.2023.12.014</pub-id><pub-id pub-id-type="medline">38246839</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ros-Arlanz&#x00F3;n</surname><given-names>P</given-names> </name><name name-style="western"><surname>Perez-Sempere</surname><given-names>A</given-names> </name></person-group><article-title>Evaluating AI competence in specialized medicine: comparative analysis of ChatGPT and neurologists in a neurology specialist examination in Spain</article-title><source>JMIR Med Educ</source><year>2024</year><month>11</month><day>14</day><volume>10</volume><fpage>e56762</fpage><pub-id pub-id-type="doi">10.2196/56762</pub-id><pub-id pub-id-type="medline">39622707</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lum</surname><given-names>ZC</given-names> </name></person-group><article-title>Can artificial intelligence pass the American Board of Orthopaedic Surgery examination? Orthopaedic residents versus ChatGPT</article-title><source>Clin Orthop Relat Res</source><year>2023</year><month>08</month><day>1</day><volume>481</volume><issue>8</issue><fpage>1623</fpage><lpage>1630</lpage><pub-id pub-id-type="doi">10.1097/CORR.0000000000002704</pub-id><pub-id pub-id-type="medline">37220190</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khan</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Sarraf</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Simpson</surname><given-names>AI</given-names> </name></person-group><article-title>Enhancements in artificial intelligence for medical examinations: a leap from ChatGPT 3.5 to ChatGPT 4.0 in the FRCS trauma &#x0026; orthopaedics examination</article-title><source>Surgeon</source><year>2025</year><month>02</month><volume>23</volume><issue>1</issue><fpage>13</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1016/j.surge.2024.11.008</pub-id><pub-id pub-id-type="medline">39613651</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yeo</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Samaan</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>WH</given-names> </name><etal/></person-group><article-title>Assessing the performance of ChatGPT in answering questions regarding cirrhosis and hepatocellular carcinoma</article-title><source>Clin Mol Hepatol</source><year>2023</year><month>07</month><volume>29</volume><issue>3</issue><fpage>721</fpage><lpage>732</lpage><pub-id pub-id-type="doi">10.3350/cmh.2023.0089</pub-id><pub-id pub-id-type="medline">36946005</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cakir</surname><given-names>H</given-names> </name><name name-style="western"><surname>Caglar</surname><given-names>U</given-names> </name><name name-style="western"><surname>Sekkeli</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Evaluating ChatGPT ability to answer urinary tract infection-related questions</article-title><source>Infect Dis Now</source><year>2024</year><month>06</month><volume>54</volume><issue>4</issue><fpage>104884</fpage><pub-id pub-id-type="doi">10.1016/j.idnow.2024.104884</pub-id><pub-id pub-id-type="medline">38460761</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Magruder</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Rodriguez</surname><given-names>AN</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>JCJ</given-names> </name><etal/></person-group><article-title>Assessing ability for ChatGPT to answer total knee arthroplasty-related questions</article-title><source>J Arthroplasty</source><year>2024</year><month>08</month><volume>39</volume><issue>8</issue><fpage>2022</fpage><lpage>2027</lpage><pub-id pub-id-type="doi">10.1016/j.arth.2024.02.023</pub-id><pub-id pub-id-type="medline">38364879</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ni</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Applications and concerns of ChatGPT and other conversational large language models in health care: systematic review</article-title><source>J Med Internet Res</source><year>2024</year><month>11</month><day>7</day><volume>26</volume><fpage>e22769</fpage><pub-id pub-id-type="doi">10.2196/22769</pub-id><pub-id pub-id-type="medline">39509695</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Briganti</surname><given-names>G</given-names> </name></person-group><article-title>How ChatGPT works: a mini review</article-title><source>Eur Arch Otorhinolaryngol</source><year>2024</year><month>03</month><volume>281</volume><issue>3</issue><fpage>1565</fpage><lpage>1569</lpage><pub-id pub-id-type="doi">10.1007/s00405-023-08337-7</pub-id><pub-id pub-id-type="medline">37991499</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ghahramani</surname><given-names>Z</given-names> </name></person-group><article-title>Probabilistic machine learning and artificial intelligence</article-title><source>Nature</source><year>2015</year><month>05</month><day>28</day><volume>521</volume><issue>7553</issue><fpage>452</fpage><lpage>459</lpage><pub-id pub-id-type="doi">10.1038/nature14541</pub-id><pub-id pub-id-type="medline">26017444</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Brain</surname><given-names>G</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Attention is all you need</article-title><access-date>2026-06-19</access-date><conf-name>31st Conference on Neural Information Processing Systems (NIPS 2017)</conf-name><conf-date>Dec 4-9, 2017</conf-date><comment><ext-link ext-link-type="uri" xlink:href="http://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf">http://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Probst</surname><given-names>P</given-names> </name><name name-style="western"><surname>Boulesteix</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Bischl</surname><given-names>B</given-names> </name></person-group><article-title>Tunability: importance of hyperparameters of machine learning algorithms</article-title><source>J Mach Learn Res</source><year>2019</year><month>03</month><day>19</day><access-date>2026-06-18</access-date><volume>20</volume><fpage>1</fpage><lpage>32</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.org/papers/volume20/18-444/18-444.pdf">https://jmlr.org/papers/volume20/18-444/18-444.pdf</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lightman</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kosaraju</surname><given-names>V</given-names> </name><name name-style="western"><surname>Burda</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Let&#x2019;s verify step by step</article-title><source>arXiv</source><comment>Preprint posted online on  May 31, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2305.20050</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><article-title>What are AI hallucinations?</article-title><source>IBM</source><access-date>2026-06-19</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ibm.com/think/topics/ai-hallucinations">https://www.ibm.com/think/topics/ai-hallucinations</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alkaissi</surname><given-names>H</given-names> </name><name name-style="western"><surname>McFarlane</surname><given-names>SI</given-names> </name></person-group><article-title>Artificial hallucinations in ChatGPT: implications in scientific writing</article-title><source>Cureus</source><year>2023</year><month>02</month><day>19</day><volume>15</volume><issue>2</issue><fpage>e35179</fpage><pub-id pub-id-type="doi">10.7759/cureus.35179</pub-id><pub-id pub-id-type="medline">36811129</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Waldo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boussard</surname><given-names>S</given-names> </name></person-group><article-title>GPTs and hallucination</article-title><source>Queue</source><year>2024</year><month>08</month><day>30</day><volume>22</volume><issue>4</issue><fpage>19</fpage><lpage>33</lpage><pub-id pub-id-type="doi">10.1145/3688007</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Cossio</surname><given-names>M</given-names> </name></person-group><article-title>A comprehensive taxonomy of hallucinations in large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 3, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.01781</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lahat</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sharif</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zoabi</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Assessing generative pretrained transformers (GPT) in clinical decision-making: comparative analysis of GPT-3.5 and GPT-4</article-title><source>J Med Internet Res</source><year>2024</year><month>06</month><day>27</day><volume>26</volume><fpage>e54571</fpage><pub-id pub-id-type="doi">10.2196/54571</pub-id><pub-id pub-id-type="medline">38935937</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Vitale</surname><given-names>J</given-names> </name><name name-style="western"><surname>Galbusera</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Is the information provided by large language models valid in educating patients about adolescent idiopathic scoliosis? An evaluation of content, clarity, and empathy: the perspective of the European Spine Study Group</article-title><source>Spine Deform</source><year>2024</year><month>11</month><day>4</day><volume>13</volume><fpage>361</fpage><lpage>372</lpage><pub-id pub-id-type="doi">10.1007/s43390-024-00955-3</pub-id><pub-id pub-id-type="medline">39495402</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ayers</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Poliak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Comparing physician and artificial intelligence chatbot responses to patient questions posted to a public social media forum</article-title><source>JAMA Intern Med</source><year>2023</year><month>06</month><day>1</day><volume>183</volume><issue>6</issue><fpage>589</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2023.1838</pub-id><pub-id pub-id-type="medline">37115527</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cadiente</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kasselman</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Pilkington</surname><given-names>B</given-names> </name></person-group><article-title>Assessing the performance of ChatGPT in bioethics: a large language model&#x2019;s moral compass in medicine</article-title><source>J Med Ethics</source><year>2024</year><month>01</month><day>23</day><volume>50</volume><issue>2</issue><fpage>97</fpage><lpage>101</lpage><pub-id pub-id-type="doi">10.1136/jme-2023-109366</pub-id><pub-id pub-id-type="medline">37973369</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Danehy</surname><given-names>T</given-names> </name><name name-style="western"><surname>Hecht</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kentis</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schechter</surname><given-names>CB</given-names> </name><name name-style="western"><surname>Jariwala</surname><given-names>SP</given-names> </name></person-group><article-title>ChatGPT performs worse on USMLE-style ethics questions compared to medical knowledge questions</article-title><source>Appl Clin Inform</source><year>2024</year><month>10</month><volume>15</volume><issue>5</issue><fpage>1049</fpage><lpage>1055</lpage><pub-id pub-id-type="doi">10.1055/a-2405-0138</pub-id><pub-id pub-id-type="medline">39209308</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vzorin</surname><given-names>GD</given-names> </name><name name-style="western"><surname>Bukinich</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Sedykh</surname><given-names>AV</given-names> </name><name name-style="western"><surname>Vetrova</surname><given-names>II</given-names> </name><name name-style="western"><surname>Sergienko</surname><given-names>EA</given-names> </name></person-group><article-title>The emotional intelligence of the GPT-4 large language model</article-title><source>Psychol Russ</source><year>2024</year><volume>17</volume><issue>2</issue><fpage>85</fpage><lpage>99</lpage><pub-id pub-id-type="doi">10.11621/pir.2024.0206</pub-id><pub-id pub-id-type="medline">39552777</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="web"><article-title>Introducing Claude</article-title><source>Anthropic</source><access-date>2023-03-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.anthropic.com/news/introducing-claude">https://www.anthropic.com/news/introducing-claude</ext-link></comment></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="web"><article-title>Claude&#x2019;s constitution</article-title><source>Anthropic</source><access-date>2023-05-09</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.anthropic.com/news/claudes-constitution">https://www.anthropic.com/news/claudes-constitution</ext-link></comment></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Bai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kadavath</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kundu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Constitutional AI: harmlessness from AI feedback</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 15, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2212.08073</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="web"><article-title>Step 2 CK sample test questions</article-title><source>United States Medical Licensing Examination (USMLE)</source><year>2026</year><access-date>2026-03-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.usmle.org/exam-resources/step-2-ck-materials/step-2-ck-sample-test-questions">https://www.usmle.org/exam-resources/step-2-ck-materials/step-2-ck-sample-test-questions</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Syed</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gadesha</surname><given-names>V</given-names> </name></person-group><article-title>What is zero-shot prompting?</article-title><source>IBM</source><access-date>2025-01-29</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ibm.com/think/topics/zero-shot-prompting">https://www.ibm.com/think/topics/zero-shot-prompting</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beckman</surname><given-names>HB</given-names> </name><name name-style="western"><surname>Frankel</surname><given-names>RM</given-names> </name></person-group><article-title>The effect of physician behavior on the collection of data</article-title><source>Ann Intern Med</source><year>1984</year><month>11</month><volume>101</volume><issue>5</issue><fpage>692</fpage><lpage>696</lpage><pub-id pub-id-type="doi">10.7326/0003-4819-101-5-692</pub-id><pub-id pub-id-type="medline">6486600</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Phillips</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Ospina</surname><given-names>NS</given-names> </name><name name-style="western"><surname>Montori</surname><given-names>VM</given-names> </name></person-group><article-title>Physicians interrupting patients</article-title><source>J Gen Intern Med</source><year>2019</year><month>10</month><volume>34</volume><issue>10</issue><fpage>1965</fpage><pub-id pub-id-type="doi">10.1007/s11606-019-05247-5</pub-id><pub-id pub-id-type="medline">31388903</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="web"><article-title>Claude&#x2019;s character</article-title><source>Anthropic</source><year>2024</year><access-date>2026-06-19</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.anthropic.com/research/claude-character">https://www.anthropic.com/research/claude-character</ext-link></comment></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Montemayor</surname><given-names>C</given-names> </name><name name-style="western"><surname>Halpern</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fairweather</surname><given-names>A</given-names> </name></person-group><article-title>In principle obstacles for empathic AI: why we can&#x2019;t replace human empathy in healthcare</article-title><source>AI Soc</source><year>2022</year><volume>37</volume><issue>4</issue><fpage>1353</fpage><lpage>1359</lpage><pub-id pub-id-type="doi">10.1007/s00146-021-01230-z</pub-id><pub-id pub-id-type="medline">34054228</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zaharia</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>J</given-names> </name></person-group><article-title>How is ChatGPT&#x2019;s behavior changing over time?</article-title><source>Harv Data Sci Rev</source><year>2023</year><month>03</month><day>13</day><volume>6</volume><issue>2</issue><pub-id pub-id-type="doi">10.1162/99608f92.5317da47</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Du</surname><given-names>W</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>X</given-names> </name><name name-style="western"><surname>Harris</surname><given-names>JC</given-names> </name><etal/></person-group><article-title>Large language models in pathology: a comparative study of ChatGPT and Bard with pathology trainees on multiple-choice questions</article-title><source>Ann Diagn Pathol</source><year>2024</year><month>12</month><volume>73</volume><fpage>152392</fpage><pub-id pub-id-type="doi">10.1016/j.anndiagpath.2024.152392</pub-id><pub-id pub-id-type="medline">39515029</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gupta</surname><given-names>M</given-names> </name><name name-style="western"><surname>Virostko</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kaufmann</surname><given-names>C</given-names> </name></person-group><article-title>Large language models in radiology: fluctuating performance and decreasing discordance over time</article-title><source>Eur J Radiol</source><year>2025</year><month>01</month><volume>182</volume><fpage>111842</fpage><pub-id pub-id-type="doi">10.1016/j.ejrad.2024.111842</pub-id><pub-id pub-id-type="medline">39581020</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Vaid</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Comparing ChatGPT and GPT-4 performance in USMLE soft skill assessments</article-title><source>Sci Rep</source><year>2023</year><month>10</month><day>1</day><volume>13</volume><issue>1</issue><fpage>16492</fpage><pub-id pub-id-type="doi">10.1038/s41598-023-43436-9</pub-id><pub-id pub-id-type="medline">37779171</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arslan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ghanem</surname><given-names>H</given-names> </name><name name-style="western"><surname>Munawar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cruz</surname><given-names>C</given-names> </name></person-group><article-title>A survey on RAG with LLMs</article-title><source>Procedia Comput Sci</source><year>2024</year><volume>246</volume><fpage>3781</fpage><lpage>3790</lpage><pub-id pub-id-type="doi">10.1016/j.procs.2024.09.178</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amugongo</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Mascheroni</surname><given-names>P</given-names> </name><name name-style="western"><surname>Brooks</surname><given-names>S</given-names> </name><name name-style="western"><surname>Doering</surname><given-names>S</given-names> </name><name name-style="western"><surname>Seidel</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name></person-group><article-title>Retrieval augmented generation for large language models in healthcare: a systematic review</article-title><source>PLOS Digit Health</source><year>2025</year><month>06</month><day>11</day><volume>4</volume><issue>6</issue><fpage>e0000877</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000877</pub-id><pub-id pub-id-type="medline">40498738</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ge</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>S</given-names> </name><name name-style="western"><surname>Owens</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Development of a liver disease-specific large language model chat interface using retrieval augmented generation</article-title><source>medRxiv</source><comment>Preprint posted online on  Nov 10, 2023</comment><pub-id pub-id-type="doi">10.1101/2023.11.10.23298364</pub-id><pub-id pub-id-type="medline">37986764</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rathje</surname><given-names>W</given-names> </name></person-group><article-title>View of learning when not to measure: theorizing ethical alignment in LLMs</article-title><source>In: Proceedings of the Seventh AAAI/ACM Conference on AI, Ethics, and Society (AIES 2024)</source><year>2024</year><volume>7</volume><fpage>1190</fpage><lpage>1199</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://ojs.aaai.org/index.php/AIES/article/view/31716/33883">https://ojs.aaai.org/index.php/AIES/article/view/31716/33883</ext-link></comment><pub-id pub-id-type="doi">10.1609/aies.v7i1.31716</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Siddarth</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lovitt</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Collective constitutional AI: aligning a language model with public input</article-title><comment>Preprint posted online on  Jun 12, 2024</comment><pub-id pub-id-type="doi">10.1145/3630106.3658979</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Upadhyay</surname><given-names>U</given-names> </name><name name-style="western"><surname>Gradisek</surname><given-names>A</given-names> </name><name name-style="western"><surname>Iqbal</surname><given-names>U</given-names> </name><name name-style="western"><surname>Dhar</surname><given-names>E</given-names> </name><name name-style="western"><surname>Li</surname><given-names>YC</given-names> </name><name name-style="western"><surname>Syed-Abdul</surname><given-names>S</given-names> </name></person-group><article-title>Call for the responsible artificial intelligence in the healthcare</article-title><source>BMJ Health Care Inform</source><year>2023</year><month>12</month><day>21</day><volume>30</volume><issue>1</issue><fpage>e100920</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2023-100920</pub-id><pub-id pub-id-type="medline">38135293</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wood</surname><given-names>G</given-names> </name><name name-style="western"><surname>Castellar</surname><given-names>EN</given-names> </name><name name-style="western"><surname>IJsselsteijn</surname><given-names>W</given-names> </name></person-group><article-title>An exploratory study into the impact of AI literacy training on anthropomorphism and trust in conversational AI</article-title><conf-name>Artificial Intelligence in HCI: 6th International Conference, AI-HCI 2025</conf-name><conf-date>Jun 22-27, 2025</conf-date><conf-loc>Gothenburg, Sweden</conf-loc><fpage>301</fpage><lpage>322</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-93415-5_18</pub-id></nlm-citation></ref></ref-list></back></article>