<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id><journal-id journal-id-type="publisher-id">mededu</journal-id><journal-id journal-id-type="index">20</journal-id><journal-title>JMIR Medical Education</journal-title><abbrev-journal-title>JMIR Med Educ</abbrev-journal-title><issn pub-type="epub">2369-3762</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v12i1e90736</article-id><article-id pub-id-type="doi">10.2196/90736</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Prompt Framing and Evidence Requirement for AI-Generated Educational Responses in Dental Education: Experimental Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Hung</surname><given-names>Man</given-names></name><degrees>MED, MBA, MSIS, MSTAT, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ward</surname><given-names>Corban</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cohen</surname><given-names>Owen</given-names></name><degrees>DDS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Marx</surname><given-names>Jacob</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Smit</surname><given-names>Zachary</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Newman</surname><given-names>Jacob</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lipsky</surname><given-names>Martin S</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>College of Dental Medicine, Roseman University of Health Sciences</institution><addr-line>10894 S. River Front Parkway</addr-line><addr-line>South Jordan</addr-line><addr-line>UT</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Family Medicine and Public Health, University of Utah</institution><addr-line>Salt Lake City</addr-line><addr-line>UT</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Education Psychology, University of Utah</institution><addr-line>Salt Lake City</addr-line><addr-line>UT</addr-line><country>United States</country></aff><aff id="aff4"><institution>Institute on Aging, Portland State University</institution><addr-line>Portland</addr-line><addr-line>OR</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Kanzow</surname><given-names>Philipp</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Abou-Bakr</surname><given-names>Asmaa</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Giannakopoulos</surname><given-names>Kostis</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Hrenczuk</surname><given-names>Marta Katarzyna</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Watanabe</surname><given-names>Plauto Christopher Aranha</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Cui</surname><given-names>Shssha</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Man Hung, MED, MBA, MSIS, MSTAT, PhD, College of Dental Medicine, Roseman University of Health Sciences, 10894 S. River Front Parkway, South Jordan, UT, 84095, United States, 1 8018781270; <email>mhung@roseman.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>6</day><month>8</month><year>2026</year></pub-date><volume>12</volume><elocation-id>e90736</elocation-id><history><date date-type="received"><day>02</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>12</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>12</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Man Hung, Corban Ward, Owen Cohen, Jacob Marx, Zachary Smit, Jacob Newman, Martin S Lipsky. Originally published in JMIR Medical Education (<ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org">https://mededu.jmir.org</ext-link>), 6.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org/">https://mededu.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mededu.jmir.org/2026/1/e90736"/><abstract><sec><title>Background</title><p>Large language models are increasingly used in health professions education; however, the role of prompt design in shaping their outputs remains poorly understood in clinical training contexts. In dentistry, where information presentation, perceived credibility, and procedural reasoning are important, the effects of instructional framing and evidence requirements on AI-generated educational responses are particularly relevant.</p></sec><sec><title>Objective</title><p>This study examined whether instructional framing and evidence requirements were associated with rater-assessed perceived factuality, tone, stance orientation, citation behavior, safety notices, hedging, and response length in responses about cavity preparation generated using GPT-5 through its web-based interface.</p></sec><sec sec-type="methods"><title>Methods</title><p>In a 2&#x00D7;2 factorial experiment, we manipulated instructional framing (patient-centered vs skill-centered) and evidence requirement (evidence-required vs no evidence) across 10 base prompt topics and 4 experimental conditions, yielding 40 outputs. Five trained raters independently coded each response for perceived factuality, confidence tone, stance orientation, hedging, citation presence, and safety notices. Response length was calculated programmatically. Interrater reliability was assessed using intraclass correlation coefficients and Fleiss &#x03BA;. Consensus measures were analyzed using factorial analyses of variance and chi-square tests.</p></sec><sec sec-type="results"><title>Results</title><p>Evidence requirement was associated with greater citation presence (19/20, 95% vs 5/20, 25%; <italic>P</italic>&#x003C;.001) and longer responses (<italic>P</italic>&#x003C;.001). It was also associated with higher rater-assessed perceived factuality in exploratory analyses (<italic>P</italic>=.02). Hedging and confidence tone showed nonsignificant patterns. Instructional framing was associated with stance orientation (<italic>P</italic>=.006) but not with response length. No refusals occurred, and safety notices were infrequent across conditions. Interrater reliability was high for citation presence but low or variable for several subjective measures, particularly perceived factuality, stance orientation and safety notices.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Prompt design was associated with differences in the presentation, structure, and orientation of large language model&#x2013;generated educational responses in dentistry. Evidence requirements increased citation inclusion and response length, whereas instructional framing was associated with the stance emphasized in the response. These findings suggest that deliberate, pedagogically aligned prompt engineering may support the design and evaluation of AI-generated content in dental education. However, the effects of prompt wording on objectively verified accuracy, clinical safety, and learning outcomes require further investigation.</p></sec></abstract><kwd-group><kwd>prompt engineering</kwd><kwd>dental education</kwd><kwd>large language models</kwd><kwd>evidence-based practice</kwd><kwd>artificial intelligence</kwd><kwd>ChatGPT</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Large language models (LLMs), such as OpenAI&#x2019;s ChatGPT series [<xref ref-type="bibr" rid="ref1">1</xref>], have rapidly transformed how information is accessed, synthesized, and applied across educational and professional domains [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. In higher education, these systems are increasingly used both formally, as tools embedded within curricula for teaching and assessment, and informally, as on-demand tutors, study aids, and feedback generators. In the health sciences, including dentistry, LLMs may help learners master complex theoretical concepts, refine procedural skills, and prepare for clinical practice. They may also assist educators with lecture development, assignment scoring, and test construction [<xref ref-type="bibr" rid="ref5">5</xref>]. However, the quality, tone, factual orientation, and presentation of LLM-generated responses can vary substantially depending on how prompts are formulated [<xref ref-type="bibr" rid="ref6">6</xref>]. This variability highlights the importance of prompt engineering, defined as the deliberate design of instructions to guide AI systems toward desired outputs [<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>Research on prompt engineering shows that even small changes in a prompt&#x2019;s wording, structure, or perspective can alter the characteristics of the resulting output [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Previous studies have demonstrated that carefully designed prompts can improve the clarity, relevance, and depth of AI-generated responses [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>], and in some contexts, enable AI-generated feedback to outperform feedback from novice humans [<xref ref-type="bibr" rid="ref11">11</xref>]. These findings suggest that prompt design is not merely a technical consideration but a pedagogically meaningful factor that influences the quality and presentation of information provided to learners. Despite these advances, most existing work has examined prompt engineering in broad academic contexts [<xref ref-type="bibr" rid="ref12">12</xref>], with limited attention to highly specialized professional education, such as dentistry, where accuracy, safety, applicability, and evidence-informed communication are particularly important.</p><p>Prompt framing refers to the perspectives or priorities embedded in prompt wording [<xref ref-type="bibr" rid="ref7">7</xref>]. Framing effects are well established in behavioral science, in which different emphases can alter attitudes and decisions even when the underlying factual content is similar [<xref ref-type="bibr" rid="ref13">13</xref>]. In LLMs, framing may shape response content, tone, stance, and elaboration [<xref ref-type="bibr" rid="ref6">6</xref>]. This possibility is relevant to dental education because patient-centered prompts may emphasize safety, comfort, and communication, whereas skill-centered prompts may emphasize technical performance or assessment outcomes.</p><p>Evidence requirements constitute another prompt-design feature by directing the model to support its responses with peer-reviewed literature. Such instructions may encourage LLMs to generate educational responses that incorporate evidence-based information [<xref ref-type="bibr" rid="ref14">14</xref>]. In dentistry, however, the accuracy and completeness of AI-generated responses require careful evaluation[<xref ref-type="bibr" rid="ref15">15</xref>]. Sourcing instructions may also increase verbosity or lead the model to generate irrelevant or fabricated references, or elicit cautionary language. Their effects on the discourse characteristics and perceived credibility of AI-generated educational responses in clinical training remain insufficiently studied.</p><p>Dental education provides a particularly relevant, high-stakes setting in which to study these prompt effects. Dental curricula combine rigorous theoretical instruction with intensive hands-on training, requiring students to develop both conceptual knowledge and precise technical skills [<xref ref-type="bibr" rid="ref16">16</xref>]. Procedures such as cavity preparation are foundational to restorative dentistry and are taught early in training [<xref ref-type="bibr" rid="ref17">17</xref>]. They require precision to support patient safety and treatment success, Errors in such procedures can have lasting clinical consequences . These characteristics make cavity preparation a useful context for examining AI-generated educational guidance. At the same time, dental students are increasingly using AI systems for rapid clarification of concepts, procedural guidance, and examination preparation [<xref ref-type="bibr" rid="ref18">18</xref>]. These trends raise important questions about how prompt wording shapes the stance, citation behavior, and perceived credibility of AI-generated educational responses.</p><p>Despite the growing integration of AI into educational practice [<xref ref-type="bibr" rid="ref19">19</xref>], few studies have examined how specific prompt features shape LLM-generated responses in dental education. This study addressed that gap by testing whether instructional framing and evidence requirements were associated with differences in GPT-5&#x2019;s responses to cavity-preparation prompts, a foundational area of preclinical dental training.</p><p>The study addressed two primary research questions: (1) Does patient-centered framing, fcompared with skill-centered framing, influence the stance, tone, and structure of GPT-5 responses? (2) Does requesting peer-reviewed evidence, compared with not requesting peer-reviewed evidence, influence citation presence, response length, hedging, safety notices, and rater-assessed perceived factuality? We also examined whether instructional framing and evidence requirement interacted to shape these response features.</p><p>We tested the following hypotheses: H1: patient-centered prompts would produce more patient welfare&#x2013;oriented responses than skill-centered prompts; H2: evidence-required prompts would produce higher rater-assessed perceived factuality scores and more frequent citation inclusion than no-evidence prompts; H3: evidence-required prompts would produce longer responses and more hedging language than no-evidence prompts; and H4: instructional framing and evidence requirement would interact such that the effect of the evidence requirement would differ between patient-centered and skill-centered prompts.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This study used a controlled, fully crossed 2&#x00D7;2 factorial experimental design to examine whether specific prompt-wording features were associated with LLM-generated outputs. Two independent variables were manipulated: instructional framing (patient-centered vs skill-centered) and evidence requirement (evidence-required vs no evidence). This design yielded 4 conditions. Each condition was applied to 10 base prompts related to cavity preparation, a foundational psychomotor and cognitive competency in preclinical dental education. The factorial design enabled within-topic comparisons across prompt variants and exploratory assessment of associations between prompt features and observable or rater-assessed response characteristics, including tone, stance, citation behavior, hedging, safety notices, and response length.</p></sec><sec id="s2-2"><title>Prompt Development and Experimental Materials</title><p>Ten base prompts were constructed to reflect diverse conceptual and procedural aspects of cavity preparation, including instrument selection, ergonomic positioning, caries removal strategies, enamel-margin design, infection-control protocols, patient communication, and clinical decision-making (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The prompts were organized into 10 topic categories representing common instructional domains in preclinical restorative dentistry. These categories were selected because they addressed foundational knowledge, psychomotor skill development, safety considerations, and communication competencies typically emphasized when novice dental learners are introduced to cavity preparation.</p><p>For each base prompt, 4 parallel variants were created by systematically combining the levels of the 2 independent variables. Patient-centered framing emphasized outcomes such as the comfort, safety, or clinical experience of patients, whereas skill-centered framing emphasized the performance, efficiency, or examination success of students. The evidence requirement factor indicated whether the prompt explicitly instructed the model to support its response with peer-reviewed dental research. In the evidence-required condition, prompts included the additional instruction, &#x201C;Please cite peer-reviewed dental research.&#x201D; This condition was intended to evaluate whether an explicit sourcing request would alter the model&#x2019;s citation behavior, academic presentation, tone, response length, and other response features. In the no-evidence condition, prompts did not instruct the model to cite literature, provide references, or justify claims using external sources. Thus, &#x201C;no evidence&#x201D; refers only to the absence of an explicit citation request; it does not imply that the topic lacked an evidence base. Apart from an identical context-reset instruction used in every trial, the 4 versions of each prompt differed only in the specific wording used to manipulate the framing and evidence components, thereby supporting conceptual comparability across conditions.</p></sec><sec id="s2-3"><title>Data Collection Procedure</title><p>Data were collected using GPT-5 as made available through the free-tier of OpenAI&#x2019;s standard web interface at the time of data collection. All responses were generated between August 19, 2025, and August 23, 2025. The model was accessed through the web-based GPT-5 interface available to the authors during data collection. No external tools, file uploads, browsing functions, or custom user-provided system instructions were used during response generation. Each prompt variant was entered as a stand-alone prompt in a new conversation. To reduce session-level carryover effects, each session instructed the model to disregard any prior context and respond only to the prompt provided in that session. This procedure was intended to minimize cross-prompt contamination and prevent access to prior examples, follow-up questions, or interaction history. Prompts were presented in a randomized sequence to reduce potential order effects, such as systematic drift in output style or adaptation across prompts.</p><p>Because data were collected through the web interface, model parameters such as temperature, top-p, hidden system instructions, safety-layer configurations, and model snapshot identifiers were neither visible nor adjustable; therefore, parameters could not be reported or controlled. For each trial, the complete text output generated by GPT-5 was copied verbatim into a structured spreadsheet, along with metadata including the prompt version, condition assignment, date and time of collection, and conversation ID. In total, 40 unique LLM-generated responses were collected&#x2014;one for each prompt-by-condition combination. These responses represented 4 experimental variants of each of the 10 base prompt topics rather than 40 conceptually independent prompt topics.</p></sec><sec id="s2-4"><title>Coding Framework and Rater Training</title><sec id="s2-4-1"><title>Overview</title><p>To evaluate the content and characteristics of the LLM-generated responses, 5 trained raters (JM, CW, OC, JN, and ZS) independently coded each of the 40 outputs. The rater group comprised dental faculty and students from an accredited US dental school. The coding rubric was developed specifically for this study through an iterative process informed by the study aims and key educational concerns in preclinical dental training. First, the research team identified outcome domains that aligned with the experimental manipulations and research questions: perceived factuality, confidence tone, stance orientation, citation presence, safety notice presence, refusal behavior, hedging, and response length. They were analyzed separately and were not treated as equally weighted components of a composite quality score. The rubric provided explicit definitions and examples for each dependent variable, as follows:</p></sec><sec id="s2-4-2"><title>Factuality (0&#x2010;3 Scale)</title><p>This measure captured raters&#x2019; perceptions of each response&#x2019;s accuracy, completeness, and clinical appropriateness. Ratings were not verified against a gold-standard reference. Consequently, this variable represents rater-assessed perceived factuality rather than objectively verified factual accuracy. Scores were defined as follows: 0=mostly inaccurate or potentially misleading content; 1=limited or incomplete accuracy with important omissions; 2=generally accurate content with minor omissions or limited detail; and 3=accurate, complete, and clinically appropriate content.</p></sec><sec id="s2-4-3"><title>Confidence Tone (0&#x2010;3 Scale)</title><p>This measure assessed the overall assertiveness or expressed certainty of the response. Scores were defined as follows: 0=highly hedged or uncertain throughout; 1=predominantly hedged with some assertiveness; 2=mostly assertive with some hedging; and 3=fully assertive and confident throughout.</p></sec><sec id="s2-4-4"><title>Stance Orientation (Categorical)</title><p>Responses were classified as patient-welfare-oriented, performance-oriented, or neutral/mixed. Patient-welfare orientation referred to responses that primarily emphasized patient safety, comfort, communication, informed consent, preservation of tooth structure, reduction of harm, or long-term patient outcomes. For example, a response that prioritized minimizing pulpal injury, reducing postoperative sensitivity, improving patient comfort, or explaining risks and benefits to support informed consent was coded as patient-welfare oriented. Performance orientation referred to responses that primarily emphasized student efficiency, grading criteria, procedural precision, examination performance, achievement of ideal preparation form, or avoidance of point deductions. For example, a response that focused on meeting dimensional tolerances, achieving ideal margins, improving speed and consistency, or performing well in practical examinations was coded as performance-oriented. Neutral/mixed orientation was assigned when a response balanced patient-welfare and performance-oriented elements or when neither orientation clearly predominated. For example, a response that provided both patient-safety guidance and examination-performance advice without prioritizing either perspective was coded as neutral/mixed.</p></sec><sec id="s2-4-5"><title>Citations Present (Binary)</title><p>This variable indicates whether the model included citation markers or bibliographic references in the response. It captured citation presence only and did not assess whether the references were authentic, whether DOIs or PMIDs were valid, whether citations were hallucinated, or whether the cited sources supported the claims made in the response.</p></sec><sec id="s2-4-6"><title>Safety Notice Flag (Binary)</title><p>This variable identifies the presence of cautionary statements, warnings, or disclaimers related to clinical risk, patient safety, or the limits of AI-generated advice.</p></sec><sec id="s2-4-7"><title>Refusal Flag (Binary)</title><p>This variable indicates whether the model declined to answer the prompt.</p></sec><sec id="s2-4-8"><title>Hedging Count (Frequency)</title><p>This variable counts hedging terms such as &#x201C;might,&#x201D; &#x201C;could,&#x201D; and &#x201C;possibly.&#x201D;</p></sec><sec id="s2-4-9"><title>Response Length (Computed Outcome)</title><p>Response length was calculated programmatically from the complete response text. Character count captured the total output size, including punctuation, formatting, citation markers, and reference-like text; word count was included as a complementary measure that may be more interpretable in educational contexts.</p><p>Before formal coding, raters completed a calibration phase using 4 pilot responses to review the rubric, discuss discrepancies, and refine operational definitions. The pilot calibration responses were excluded from the final analytic dataset and were not included in the statistical analyses. After calibration, raters independently coded the 40 study responses while blinded to condition assignments. The variable coding is presented in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec></sec><sec id="s2-5"><title>Data Consolidation and Consensus Procedures</title><p>After coding, a multistep process was used to generate consensus values for each dependent variable. For ordinal variables (rater-assessed perceived factuality and confidence tone), mean ratings across the 5 raters were calculated. For binary outcomes (citations, safety notices, and refusals), majority agreement served as the consensus rule. Stance orientation required additional adjudication: when raters&#x2019; classifications resulted in a tie, the tie was resolved using one author&#x2019;s rating (OC&#x2019;s), as specified a priori, because OC has greater clinical experience and familiarity with preclinical dental education. This role was intended to provide an experienced clinical perspective in cases of disagreement. To reduce the potential for individual-level bias, the tie-breaking procedure was defined in advance, all raters used a structured rating rubric, and all raters completed calibration using standardized definitions before formal evaluation. Response length was calculated directly from the raw text to ensure consistency across responses.</p></sec><sec id="s2-6"><title>Interrater Reliability Assessment</title><p>Interrater reliability was evaluated to assess the consistency of the coding process. For ordinal outcomes, 2-way random-effects intraclass correlation coefficients [ICC (2,1) and ICC (2,k)] assessed single-rater and average-rater reliability. For nominal outcomes, Fleiss &#x03BA; quantified multirater agreement. Reliability varied across measures, with excellent agreement for citation presence, fair average-rater reliability for confidence tone, good averae-rater reliablity for hedging counts, and low or negative agreement for perceived factuality, safety notices, and stance orientation. Negative ICC or &#x03BA; values were interpreted as indicating no reliable agreement beyond chance. These reliability estimates informed the interpretation of consensus-based measures in subsequent statistical analyses.</p></sec><sec id="s2-7"><title>Statistical Analysis</title><p>All analyses were conducted using the consensus-coded dataset. The analytic dataset included 40 model-generated responses derived from 10 base prompt topics, with each topic represented under 4 experimental conditions defined by instructional framing and evidence requirements.</p><p>For continuous and ordinal dependent variables, including rater-assessed perceived factuality, confidence tone, response length, and hedging count, 2-way factorial ANOVAs were conducted to evaluate the main effects of instructional framing and evidence requirement and their interaction. Each analysis included instructional framing, evidence requirement, and the framing-by-evidence interaction as fixed factors. The prespecified analyses treated the 40 generated responses as the analytic observations, yielding 1 numerator and 36 denominator df for each main and interaction effect. Effect sizes were reported as partial eta-squared (&#x03B7;&#x00B2;).</p><p>For categorical dependent variables, including citation presence, safety-notice presence, and stance orientation, chi-square tests of independence were conducted to evaluate associations with instructional framing and evidence requirements. Cram&#x00E9;r <italic>V</italic> was reported as the corresponding effect-size measure.</p><p>The refusal variable was excluded from inferential testing because all responses answered the corresponding prompt, resulting in no variability. Statistical significance was evaluated using a 2-sided threshold of <italic>P</italic>&#x003C;.05. All analyses used consensus values derived from the 5 raters.</p></sec><sec id="s2-8"><title>Ethical Considerations</title><p>This study evaluated GPT-5 generated responses to simulated educational prompts and did not involve patients, biological specimens, or identifiable private information. The source responses analyzed were generated by GPT-5 rather than by human participants, Accordingly, the study did not involve human subjects as defined under the US Common Rule (45 CFR &#x00A7;46.102(e)(1)); therefore, institutional review board review and approval were not required . No protected health information subject to the Health Insurance Portability and Accountability Act was used.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Interrater Reliability</title><p>Before conducting the main analyses, interrater reliability was assessed across the 5 independent raters (JM, CW, OC, JN, and ZS). Reliability was calculated using ICCs for ordinal and count-based measures and Fleiss &#x03BA; for categorical variables. Agreement varied across measures (<xref ref-type="table" rid="table1">Table 1</xref>). Citation presence demonstrated excellent agreement (&#x03BA;=0.84), indicating that raters were highly consistent in identifying whether a response included references. Hedging counts showed good average-rater reliability [ICC (2,k)=0.66] and poor-to-fair single-rater reliability [ICC (2,1)=0.28]. Confidence tone ratings showed fair average-rater reliability [ICC (2,k)=0.45]. Stance orientation exhibited slight agreement (&#x03BA;=0.17), indicating that raters differed in classifying the underlying stance of some responses. Safety notice flags showed no meaningful agreement (&#x03BA;=&#x2212;0.02), as did perceived factuality ratings [ICC (2,k)=&#x2212;0.16], indicating substantial variability among raters in judging these criteria. Finally, all raters agreed that every GPT-5 output answered the corresponding prompt, resulting in unanimous coding for response completion across all 40 items.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Interrater reliability across the 5 raters (N=40 responses).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Measure</td><td align="left" valign="bottom">Interrater reliability type</td><td align="left" valign="bottom">Interrater reliability value</td><td align="left" valign="bottom">Interpretation</td></tr></thead><tbody><tr><td align="left" valign="top">Factuality (0&#x2010;3)</td><td align="left" valign="top">ICC<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> (2,1)</td><td align="left" valign="top">&#x2212;0.029</td><td align="left" valign="top">Poor reliability: individual raters did not consistently agree in their perceived factuality ratings.</td></tr><tr><td align="left" valign="top">Factuality (0&#x2010;3)</td><td align="left" valign="top">ICC (2,k)</td><td align="left" valign="top">&#x2212;0.161</td><td align="left" valign="top">Poor reliability: even aggregated ratings showed no reliable agreement; factuality findings should therefore be treated as exploratory.</td></tr><tr><td align="left" valign="top">Confidence tone (0&#x2010;3)</td><td align="left" valign="top">ICC (2,1)</td><td align="left" valign="top">0.138</td><td align="left" valign="top">Poor reliability: single raters were not aligned in how they judged confidence tone.</td></tr><tr><td align="left" valign="top">Confidence tone (0&#x2010;3)</td><td align="left" valign="top">ICC (2,k)</td><td align="left" valign="top">0.445</td><td align="left" valign="top">Fair reliability: averaging all 5 raters produced moderate consistency, suggesting partial but imperfect agreement on confidence tone.</td></tr><tr><td align="left" valign="top">Hedging count</td><td align="left" valign="top">ICC (2,1)</td><td align="left" valign="top">0.278</td><td align="left" valign="top">Poor-to-fair reliability: individual raters agreed on how often hedging language appeared, but differences remained.</td></tr><tr><td align="left" valign="top">Hedging count</td><td align="left" valign="top">ICC (2,k)</td><td align="left" valign="top">0.658</td><td align="left" valign="top">Good reliability: when all raters&#x2019; counts were averaged, agreement improved substantially, indicating consistent group-level scoring.</td></tr><tr><td align="left" valign="top">Stance orientation</td><td align="left" valign="top">Fleiss &#x03BA;</td><td align="left" valign="top">0.174</td><td align="left" valign="top">Slight agreement: raters only minimally agreed on whether responses were patient-oriented, performance-oriented, or mixed.</td></tr><tr><td align="left" valign="top">Citations present</td><td align="left" valign="top">Fleiss &#x03BA;</td><td align="left" valign="top">0.844</td><td align="left" valign="top">Almost perfect agreement: raters nearly always agreed on whether citations were included, showing strong consistency.</td></tr><tr><td align="left" valign="top">Safety notice flag</td><td align="left" valign="top">Fleiss &#x03BA;</td><td align="left" valign="top">&#x2212;0.016</td><td align="left" valign="top">No meaningful agreement: raters did not consistently identify safety notices, limiting interpretation of this outcome.</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>ICC: intraclass correlation coefficient.</p></fn></table-wrap-foot></table-wrap><p>Given these reliability findings, consensus values were used to summarize response features across conditions. However, outcomes with low or negative reliability, particularly perceived factuality and safety notices, were interpreted only as exploratory and were not used to support conclusions about objective accuracy or clinical safety.</p></sec><sec id="s3-2"><title>Descriptive Statistics</title><p><xref ref-type="table" rid="table2">Table 2</xref> summarizes descriptive statistics for all dependent variables across the 4 experimental conditions. The sample included 40 GPT-5 responses generated from 10 base prompt topics crossed with 4 prompt conditions. Mean rater-assessed perceived factuality scores were high across conditions, ranging from 2.58 to 2.80 on the 0 to 3 scale. Mean confidence tone ratings indicated generally assertive outputs and ranged from 2.22 to 2.60. Response length varied by evidence condition. Evidence-required prompts produced notably longer outputs (approximately 4900-5300 characters; 724-774 words) than did no-evidence prompts (approximately 3100-3200 characters; 449-459 words). Hedging was infrequent, with mean counts ranging from 1.70 to 3.28 per response.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Descriptive statistics for dependent variables by instructional framing and evidence requirement.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variable</td><td align="left" valign="bottom">Overall (N=40)</td><td align="left" valign="bottom">Patient-centered, no evidence (n=10)</td><td align="left" valign="bottom">Patient-centered, evidence-required (n=10)</td><td align="left" valign="bottom">Skill-centered, no evidence (n=10)</td><td align="left" valign="bottom">Skill-centered, evidence-required (n=10)</td></tr></thead><tbody><tr><td align="left" valign="top">Rater-assessed perceived factuality, mean (SD)</td><td align="left" valign="top">2.71 (0.19)</td><td align="left" valign="top">2.58 (0.15)</td><td align="left" valign="top">2.76 (0.18)</td><td align="left" valign="top">2.70 (0.17)</td><td align="left" valign="top">2.80 (0.19)</td></tr><tr><td align="left" valign="top">Confidence tone, mean (SD)</td><td align="left" valign="top">2.47 (0.35)</td><td align="left" valign="top">2.60 (0.33)</td><td align="left" valign="top">2.22 (0.36)</td><td align="left" valign="top">2.50 (0.29)</td><td align="left" valign="top">2.54 (0.34)</td></tr><tr><td align="left" valign="top">Length (characters), mean (SD)</td><td align="left" valign="top">4111 (1358)</td><td align="left" valign="top">3109 (1251)</td><td align="left" valign="top">4936 (714)</td><td align="left" valign="top">3148 (346)</td><td align="left" valign="top">5258 (1187)</td></tr><tr><td align="left" valign="top">Length (words), mean (SD)</td><td align="left" valign="top">602 (200)</td><td align="left" valign="top">449 (183)</td><td align="left" valign="top">724 (96)</td><td align="left" valign="top">459 (57)</td><td align="left" valign="top">774 (172)</td></tr><tr><td align="left" valign="top">Hedging count, mean (SD)</td><td align="left" valign="top">2.27 (1.43)</td><td align="left" valign="top">1.70 (1.35)</td><td align="left" valign="top">3.28 (1.45)</td><td align="left" valign="top">2.08 (1.25)</td><td align="left" valign="top">2.02 (1.34)</td></tr><tr><td align="left" valign="top">Citations present, n (%)</td><td align="left" valign="top">24 (60)</td><td align="left" valign="top">3 (30)</td><td align="left" valign="top">9 (90)</td><td align="left" valign="top">2 (20)</td><td align="left" valign="top">10 (100)</td></tr><tr><td align="left" valign="top">Safety notices, n (%)</td><td align="left" valign="top">5 (13)</td><td align="left" valign="top">4 (40)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1 (10)</td></tr><tr><td align="left" valign="top">Refusals, n (%)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td></tr></tbody></table></table-wrap><p>Overall, 24 out of 40 (60%) responses included citations, but citation presence differed sharply by evidence condition. Nineteen out of 20 (95%) evidence-required responses contained citations, compared with 5 out of 20 (25%) no-evidence responses. Safety notices were infrequent, occurring in 5 out of 40 (13%) responses, with the highest frequency observed in the patient-centered, no-evidence condition (4/10, 40%). No refusals occurred in any condition. Stance orientation differed by framing: skill-centered prompts overwhelmingly produced performance-oriented responses, whereas patient-centered prompts produced a more heterogeneous distribution across stance categories.</p></sec><sec id="s3-3"><title>Effects of Instructional Framing and Evidence Requirements</title><p>A series of 2&#x00D7;2 factorial ANOVAs and chi-square tests examined the main and interaction effects of framing (patient-centered vs skill-centered) and evidence requirement (evidence-required vs. no-evidence) on the dependent variables.</p><sec id="s3-3-1"><title>Rater-Assessed Perceived Factuality</title><p>Evidence requirement was associated with higher mean rater-assessed perceived factuality scores, <italic>F</italic><sub>1,36</sub>=6.53, <italic>P</italic>=.02, &#x03B7;&#x00B2;=0.154 (<xref ref-type="table" rid="table3">Table 3</xref>). Responses generated from evidence-required prompts had higher mean scores (mean 2.78, SD 0.18) than responses generated without an evidence instruction (mean 2.64, SD 0.17). The main effect of framing was not significant, <italic>F</italic><sub>1,36</sub>=2.13, <italic>P</italic>=.15, and the interaction between framing and evidence requirement was also nonsignificant, <italic>F</italic><sub>1,36</sub>=0.53, <italic>P</italic>=.47. Both framing conditions showed higher perceived factuality scores when evidence was required. Because interrater reliability for this outcome was negative, the result should be interpreted as exploratory.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Summary of statistical tests for the main and interaction effects of instructional framing and evidence requirements on dependent variables<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dependent variable</td><td align="left" valign="bottom">Test type</td><td align="left" valign="bottom">Framing effect</td><td align="left" valign="bottom">Evidence requirement effect</td><td align="left" valign="bottom">Interaction effect</td></tr></thead><tbody><tr><td align="left" valign="top">Stance orientation</td><td align="left" valign="top">Chi-square</td><td align="left" valign="top">&#x03C7;<sup>2</sup> (<italic>df</italic>)=10.19 (2), <italic>P</italic>=.006, <italic>V</italic>=0.50</td><td align="left" valign="top">&#x03C7;<sup>2</sup> (<italic>df</italic>)=2.91 (2), <italic>P</italic>=.23, <italic>V</italic>=0.27</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Rater-assessed perceived factuality (0&#x2010;3)</td><td align="left" valign="top">ANOVA</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=2.13 (1,36), <italic>P</italic>=.15, &#x03B7;&#x00B2;=0.056</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=6.53 (1,36), <italic>P</italic>=.02, &#x03B7;&#x00B2;=0.154</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=0.53 (1,36), <italic>P</italic>=.47, &#x03B7;&#x00B2;=0.015</td></tr><tr><td align="left" valign="top">Confidence tone (0&#x2010;3)</td><td align="left" valign="top">ANOVA</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=1.12 (1,36), <italic>P</italic>=.30, &#x03B7;&#x00B2;=0.030</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=2.67 (1,36), <italic>P</italic>=.11, &#x03B7;&#x00B2;=0.069</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=4.07 (1,36), <italic>P</italic>=.05, &#x03B7;&#x00B2;=0.102</td></tr><tr><td align="left" valign="top">Length (characters)</td><td align="left" valign="top">ANOVA</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=0.35 (1,36), <italic>P</italic>=.56, &#x03B7;&#x00B2;=0.010</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=43.07 (1,36), <italic>P</italic>&#x003C;.001, &#x03B7;&#x00B2;=0.545</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=0.23 (1,36), <italic>P</italic>=.63, &#x03B7;&#x00B2;=0.006</td></tr><tr><td align="left" valign="top">Mean hedging count</td><td align="left" valign="top">ANOVA</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=1.06 (1,36), <italic>P</italic>=.31, &#x03B7;&#x00B2;=0.029</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=3.16 (1,36), <italic>P</italic>=.08, &#x03B7;&#x00B2;=0.08</td><td align="left" valign="top"><italic>F</italic> test (<italic>df</italic>)=3.68 (1,36), <italic>P</italic>=.06, &#x03B7;&#x00B2;=0.093</td></tr><tr><td align="left" valign="top">Citations present</td><td align="left" valign="top">Chi-square test</td><td align="left" valign="top">&#x03C7;<sup>2</sup> (<italic>df</italic>)=0.00 (1), <italic>P</italic>&#x003E;.99, <italic>V</italic>=0.000</td><td align="left" valign="top">&#x03C7;<sup>2</sup> (<italic>df</italic>)=17.60 (1), <italic>P</italic>&#x003C;.001, <italic>V</italic>=0.663</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Safety notices</td><td align="left" valign="top">Chi-square test</td><td align="left" valign="top">&#x03C7;<sup>2</sup> (<italic>df</italic>)=0.91 (1), <italic>P</italic>=.34, <italic>V</italic>=0.151</td><td align="left" valign="top">&#x03C7;<sup>2</sup> (<italic>df</italic>)=0.91 (1), <italic>P</italic>=.34, <italic>V</italic>=0.151</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>&#x03B7;&#x00B2;= eta-squared for ANOVA results; <italic>V</italic>=Cram&#x00E9;r <italic>V</italic> for chi-square results.</p></fn><fn id="table3fn2"><p><sup>b</sup>Not tested.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3-2"><title>Confidence Tone</title><p>Neither framing nor evidence requirement had a statistically significant main effect on confidence-tone ratings (<xref ref-type="table" rid="table3">Table 3</xref>). The framing-by-evidence interaction was also not statistically significant, <italic>F</italic><sub>1,36</sub>=4.07, <italic>P</italic>=.05, &#x03B7;&#x00B2;=0.102. Descriptively, adding an evidence requirement lowered the mean confidence tone in the patient-centered condition (mean 2.60, SD 0.33 to mean 2.22, SD 0.36) but had little effect in the skill-centered condition (mean 2.50, SD 0.29 to mean 2.54, SD 0.34). This suggests that patient-centered prompts may elicit a more cautious tone when combined with an explicit demand for citations.</p></sec><sec id="s3-3-3"><title>Response Length</title><p>Evidence requirement had a large main effect on response length, <italic>F</italic><sub>1,36</sub>=43.07, <italic>P</italic>&#x003C;.001, &#x03B7;&#x00B2;=0.545 (<xref ref-type="table" rid="table3">Table 3</xref>). Evidence-required outputs were substantially longer (mean 5097, SD 967 characters) than no-evidence outputs (mean 3129, SD 894 characters), an increase of approximately 1968 characters, or 63%. Neither the main effect of instructional framing nor the interaction was statistically significant.</p></sec><sec id="s3-3-4"><title>Hedging</title><p>Neither framing nor evidence requirement had a statistically significant effect on hedging counts (<xref ref-type="table" rid="table3">Table 3</xref>). Evidence-required prompts produced a numerically higher mean hedging count (mean 2.65, SD 1.51) than no-evidence prompts (mean 1.89, SD 1.28), <italic>F</italic><sub>1,36</sub>=3.16, <italic>P</italic>=.08, &#x03B7;&#x00B2;=0.080. The framing-by-evidence interaction was also not statistically significant, <italic>F</italic><sub>1,36</sub>=3.68, <italic>P</italic>=.06, &#x03B7;&#x00B2;=0.093.</p></sec></sec><sec id="s3-4"><title>Categorical Outcomes</title><sec id="s3-4-1"><title>Citations Present</title><p>Chi-square analysis demonstrated a strong association between evidence requirement and citation presence, <italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=17.60, <italic>P</italic>&#x003C;.001, <italic>V</italic>=0.663 (<xref ref-type="table" rid="table3">Table 3</xref>). Evidence-required prompts produced citations in 19 out of 20 (95%) responses, whereas no-evidence prompts produced citations in 5 out of 20 (25%) responses. Instructional framing was not associated with citation presence, <italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=0.00, <italic>P</italic>&#x003E;.99.</p></sec><sec id="s3-4-2"><title>Safety Notices</title><p>Safety notices were rare. Neither instructional framing nor evidence requirement was significantly associated with safety notice frequency; for both tests, <italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=0.91, <italic>P</italic>=.34 (<xref ref-type="table" rid="table3">Table 3</xref>). Although the difference was not statistically significant, the patient-centered, no-evidence condition had the highest frequency of safety notices (4/10 responses).</p></sec><sec id="s3-4-3"><title>Refusals</title><p>All raters agreed that no refusals occurred in any condition (<xref ref-type="table" rid="table3">Table 3</xref>). GPT-5 answered every prompt, including those that requested citations.</p></sec><sec id="s3-4-4"><title>Stance Orientation</title><p>Instructional framing was significantly associated with stance orientation, <italic>&#x03C7;</italic>&#x00B2;<sub>2</sub>=10.19, <italic>P</italic>=.006, <italic>V</italic>=0.50 (<xref ref-type="table" rid="table3">Table 3</xref>). Skill-centered prompts elicited performance-oriented responses in 19 out of 20 cases, whereas patient-centered prompts produced a more heterogeneous distribution across stance categories. This result was consistent with the intended framing manipulation. However, evidence requirement was not significantly associated with stance orientation, <italic>&#x03C7;</italic>&#x00B2;<sub>2</sub>=2.91, <italic>P</italic>=.23.</p></sec></sec><sec id="s3-5"><title>Summary of Findings</title><p>Overall, evidence requirements were associated with higher rater-assessed perceived factuality scores, greater citation presence, and longer responses. Instructional framing was associated with stance orientation, with skill-centered prompts more often producing performance-oriented responses. Because perceived factuality showed negative interrater reliability, its association with evidence requirements should be considered exploratory and should not be interpreted as evidence of objective factual accuracy. Descriptive differences in hedging and confidence tone did not reach statistical significance. No refusals occurred, and safety notices were infrequent across conditions. Together, these findings indicate that prompt design elements were associated with distinct observable and rater-assessed features of AI-generated dental education responses, particularly citation behavior, response length, and stance orientation.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Summary</title><p>This study examined whether instructional framing and evidence requirements were associated with differences in the rater-assessed perceived factuality, tone, stance, citation behavior, safety notices, hedging, response length and other presentation features of GPT-5 responses in dental education. By systematically manipulating these prompt characteristics in a 2&#x00D7;2 factorial design and applying multirater coding to the resulting outputs, the study contributes to a growing understanding of how LLM-generated educational responses may vary with prompt wording. The findings extend prior work on the potential of LLMs in instructional design and assessment in health professions education while underscoring the need for careful evaluation, transparent reporting, and appropriate governance [<xref ref-type="bibr" rid="ref20">20</xref>].</p></sec><sec id="s4-2"><title>Instructional Framing and Response Stance</title><p>One of the clearest findings concerned stance orientation. Instructional framing was associated with the stance expressed in GPT-5&#x2019;s responses. Skill-centered prompts overwhelmingly yielded performance-oriented outputs, whereas patient-centered prompts generated a broader distribution of stance categories. This pattern is consistent with prompt-framing research, showing that models can adopt the priorities, goals, and perspectives embedded in the prompt itself [<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>However, the observed asymmetry&#x2014;skill-centered prompts strongly directed response stance, whereas patient-centered prompts produced a less consistent shift toward patient welfare&#x2014;raises questions about how this model represented competing instructional priorities. In this dataset, explicitly skill-centered wording was consistently associated with performance-oriented responses, but patient-centered wording did not uniformly elicit patient-welfare&#x2013;oriented responses. In clinical education domains such as dentistry, where patient welfare is foundational [<xref ref-type="bibr" rid="ref22">22</xref>], educators should therefore avoid assuming that patient-centered priorities will be consistently emphasized unless they are clearly specified and reinforced. This concern is consistent with emerging dental education literature suggesting that AI tools may shape students&#x2019; professional identity formation and reflective practice depending on how they are positioned and scaffolded in curricula [<xref ref-type="bibr" rid="ref23">23</xref>].</p></sec><sec id="s4-3"><title>Evidence Requirements on Response Characteristics</title><p>The evidence requirement was associated with several observable response features. Requiring peer-reviewed citations substantially increased citation presence: 19 out of 20 (95%) evidence-required outputs contained citations, compared with 5 out of 20 (25%) no-evidence outputs. This finding aligns with research indicating that LLMs respond strongly to sourcing instructions and often expand on their explanations or include reference-like material when prompted to do so [<xref ref-type="bibr" rid="ref24">24</xref>]. However, the result reflects a change in citation behavior and reference-like formatting , not evidence of improved scientific rigor. Citation presence was coded only as a structural feature. The study did not verify whether references were authentic, whether DOIs or PMIDs were valid, whether citations were hallucinated, or whether the cited sources accurately supported the model&#x2019;s claims. This distinction is important because multiple studies have documented fabricated or erroneous references in GPT-5-generated biomedical content [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Accordingly, evidence-required prompting may increase the appearance of evidence-based communication, but citation inclusion alone should not be treated as evidence of source quality, citation validity, or factual support.</p><p>Evidence-required prompts also elicited substantially longer outputs, increasing from a mean of approximately 3129 (SD 894) characters in the no-evidence condition to 5097 (SD 967) characters in the evidence-required condition. This difference corresponds to an increase of approximately 1968 characters, or 63%. The expansion may reflect the model&#x2019;s tendency to imitate academic writing conventions when asked to cite peer-reviewed research. For educators, this creates a practical trade-off: evidence requests may produce more source-like formatting and elaboration, but they may also increase verbosity and potentially obscure key instructional points for novice learners.</p><p>Interestingly, evidence requirements showed a trend toward greater hedging, particularly in the patient-centered condition. This pattern may indicate that when the model is asked to provide evidence within a patient-oriented framing, it adopts more cautious language. Such caution may reflect the model&#x2019;s sensitivity to clinical risk, evidentiary uncertainty, or the constraints of responding without direct access to verified databases [<xref ref-type="bibr" rid="ref27">27</xref>]. Although this interaction did not reach statistical significance, the observed direction suggests that evidence requirements in clinical domains may influence not only response structure but also communicative tone.</p></sec><sec id="s4-4"><title>Confidence Tone and the Complexities of AI Self-Presentation</title><p>Confidence tone ratings were high across conditions, reflecting the fluent and assertive style characteristic of contemporary LLM outputs. Neither the main effect nor the framing-by-evidence interaction was statistically significant. Therefore, this result should be interpreted as an exploratory observation rather than as evidence of a reliable effect.</p><p>Descriptively, confidence tone was lower in the patient-centered, evidence-required condition than in the patient-centered, no-evidence condition, whereas confidence tone was similar across evidence conditions for skill-centered prompts. This pattern may warrant further study, but the present findings do not establish that patient-centered framing, citation requests, or their combination systematically influence the certainty or caution expressed in AI-generated clinical education responses.</p></sec><sec id="s4-5"><title>Minimal Effects on Safety Notices and Refusals</title><p>Safety notices occurred rarely and inconsistently across conditions, with no statistically significant differences between evidence-required and no-evidence prompts. The cavity-preparation topics may not have elicited explicit safety disclaimers, or the prompts may have been interpreted as educational rather than as requests for directive clinical advice. Because safety notice coding showed no meaningful interrater agreement, these findings should not be interpreted as evidence that prompt design did or did not influence clinical safety.</p><p>Similarly, no refusals occurred in any condition; GPT-5 answered all prompts, including those requesting citations. Although high response compliance may improve usability, it also raises a potential concern: models may provide reference-like material even when those references are not externally verified. This may result in hallucinated or unsupported citations. Given documented concerns about fabricated citations in biomedical contexts, dental educators may wish to request identifiers such as PMIDs or DOIs and independently verify them in PubMed, Crossref, or publisher&#x2019;s website as part of AI literacy training [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>].</p></sec><sec id="s4-6"><title>Implications for Dental Education and Clinical Training</title><p>These findings are most directly relevant to how dental educators design, evaluate, and scaffold prompts when using AI-generated content in instructional settings. Because this study examined model outputs rather than learner outcomes, its implications concern the design of AI-supported educational materials rather than the direct effects on student learning or clinical performance.</p><p>First, the association between instructional framing and response stance suggests that educators should craft prompts intentionally to reflect the instructional priorities they wish to emphasize, whether patient-centered reasoning, procedural efficiency, reflective analysis, or examination preparation. Because patient-centered prompts did not uniformly produce patient-welfare&#x2013;oriented responses, educators should not assume that patient priorities will be consistently emphasized unless those priorities are explicitly stated and reinforced.</p><p>Second, the effects of evidence requirements must be balanced against their drawbacks. Although evidence-required prompts were associated with higher rater-assessed perceived factuality scores in exploratory analyses and substantially increased citation presence, they also produced longer responses and may introduce pseudoacademic formatting that could confuse learners or obscure essential concepts. An evidence requirement may also alter the apparent certainty of a response, although this study did not establish such an effect. These complexities highlight the need for explicit instructions on how students should interpret, verify, and critically evaluate AI-generated citations.</p><p>Third, the low or variable interrater reliability observed for several measures, including perceived factuality and safety notices, underscores the difficulty of evaluating AI outputs consistently, even among trained raters. Consensus scoring does not eliminate this measurement uncertainty. Educators should consider scaffolded training to prepare students and faculty to assess the rigor, clinical appropriateness, and limitations of AI-generated explanations. As AI becomes more integrated into health professions education, the ability to critically appraise model outputs is likely to become an essential professional competency.</p></sec><sec id="s4-7"><title>Theoretical Contributions to Prompt Engineering Research</title><p>This study contributes to the prompt engineering literature by showing that framing and evidence instructions may be associated with different dimensions of LLM-generated responses. Framing was primarily associated with response perspective, particularly stance orientation, whereas evidence instructions were associated with response structure and citation behavior; an exploratory analysis also suggested an association with perceived factuality. The pattern suggests that different prompt components may influence distinct aspects of AI-generated communication.</p><p>The study did not identify statistically significant interactions for most variables. Instead, 2 prompt factors were associated with different response features: framing was associated with rhetorical orientation, whereas evidence requirements were associated with elaboration and citation inclusion. This distinction may guide future work examining how prompt components interact to shape various dimensions of AI-generated pedagogical content.</p></sec><sec id="s4-8"><title>Limitations</title><p>Several limitations warrant consideration. First, stance orientation exhibited low reliability across raters, suggesting that coding frameworks for complex discourse characteristics require further refinement. Perceived factuality and safety notices also showed low or negative interrater reliability, indicating that these domains were difficult to apply consistently. Although perceived factuality was rated using a structured rubric by trained raters, it did not represent objectively verified clinical accuracy. Therefore, it should be interpreted as rater-assessed perceived accuracy and completeness rather than as an objective measure of factual truth. Similarly, the presence of a safety notice should not be interpreted as a robust indicator of clinical safety. Findings involving these lower-reliability outcomes should be considered exploratory, and consensus scoring does not remove the underlying measurement uncertainty. Intrarater reliability was also not assessed because the study design did not include a repeat-rating phase in which the raters recoded the same responses after a defined interval. Consequently, this study could not determine whether individual raters applied the rubric consistently over time, particularly for interpretive domains such as perceived factuality, safety notices, and stance orientation.</p><p>Second, although the factorial design included 40 generated responses, these responses were derived from 10 base prompt topics crossed with 4 experimental conditions. Thus, the study&#x2019;s conceptual breadth is better understood as 10 prompt topics rather than 40 fully independent prompts. The statistical analyses treated the 40 responses as analytic observations and did not account for the grouping of 4 prompt variants within each base topic. Because responses derived from the same topic shared underlying subject matter, the observations may not have been independent. Failing to account for this clustering may have produced SEs that did not reflect within-topic dependence and may have overstated the precision of the inferential results. In addition, the small number of base topics and sparse categorical events limited statistical power and generalizability. Accordingly, the findings should be interpreted as exploratory, descriptive patterns within this structured prompt set rather than as definitive evidence of broadly generalizable effects.</p><p>Third, reliance on the free-tier GPT-5 flagship model limits generalizability to other models, paid versions, or future system updates, which may differ in safety behavior, formatting tendencies, citation behavior, or response style. Because all outputs were generated by a single model through one interface, the findings reflect GPT-5 behavior at the time of data collection rather than prompt effects across all LLM systems. Although each prompt was administered in a new conversation with instructions to disregard prior context, this procedure reduced session-level carryover but did not make the outputs independent of the underlying model system.</p><p>Relatedly, the study used the publicly accessible GPT-5 web interface rather than an application programming interface with fixed, reportable generation parameters. Consequently, technical details such as temperature, top-p, hidden system prompts, safety-layer settings, alignment updates, and model snapshot identifiers were unavailable and could not be controlled. Because LLM systems are updated over time, the exact outputs generated in this study may not be fully reproducible in future sessions. To improve transparency, the study reports the data collection procedure in detail and provides sample prompts and responses in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>Fourth, citation presence was coded as a structural feature but was not equivalent to citation validity, evidence quality, or the correct interpretation of the cited literature. The study did not verify DOIs or PMIDs, confirm whether each cited reference corresponded to an authentic publication, assess hallucinated references, or evaluate whether the cited literature supported the model&#x2019;s claims.</p><p>Finally, although cavity preparation is a foundational topic in dental training, the study&#x2019;s narrow clinical-procedural scope limits the generalizability. Because the study examined a single domain, the observed prompt effects may not extend to other areas of dental education, including diagnosis, treatment planning, prevention, patient communication, ethics, or other restorative procedures.</p></sec><sec id="s4-9"><title>Future Directions</title><p>Future research should investigate how students and faculty use and interpret AI-generated explanations. Studies could examine whether learners mistake unverified or fabricated citations as genuine evidence, how confidence tone influences trust and study behavior, and whether stance orientation affects learning outcomes such as conceptual understanding or clinical reasoning. Future studies should also use larger and more diverse prompt sets across multiple dental education domains, including topics with different levels of evidentiary support. Incorporating validated reference answers, expert-consensus standards, or evidence summaries would allow for more rigorous assessment of factual accuracy and clinical appropriateness.</p><p>Additional work should refine and validate coding rubrics for evaluating LLM-generated dental education content. Future studies should assess both interrater and intrarater reliabilities, with intrarater reliability evaluated by having raters recode a subset of responses after a defined interval. This approach would help determine whether individual raters apply the rubric consistently over time, particularly for interpretive domains such as rater-assessed perceived factuality, stance orientation, safety notices, and hedging.</p><p>Experiments involving interactive, multiturn dialogue may also reveal more dynamic framing effects as the model adapts to user responses. More sophisticated prompt engineering strategies, such as structured reasoning scaffolds, self-check prompts, reference-verification prompts, and citation-validity checks, could also be tested to determine whether they better align LLM outputs with educational and ethical goals.</p></sec><sec id="s4-10"><title>Conclusion</title><p>This study found that prompt framing and evidence instructions were associated with systematic differences in the discourse characteristics of GPT-5 responses related to cavity preparation. Evidence requirements were associated with greater citation inclusion, longer responses, and higher rater-assessed perceived factuality scores; instructional framing was associated primarily with response orientation. These findings indicate that prompt wording can influence how LLM-generated educational responses are presented and perceived. As generative AI use expands in dental education, deliberate prompt design may help educators better align AI-generated content with intended instructional objectives. However, objective accuracy, citation validity, clinical safety, and effects on learning require direct evaluation.</p></sec></sec></body><back><ack><p>The authors thank the Analytic Galaxy and Roseman University Clinical Outcomes Research and Education Center for conducting the statistical analyses.</p><p>The source responses evaluated in this study were generated through simulations performed using ChatGPT-5.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The data supporting the findings of this study are available in the supplementary materials accompanying this article, including sample prompts and GPT-5 responses.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">HIPAA</term><def><p>Health Insurance Portability and Accountability Act</p></def></def-item><def-item><term id="abb2">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><source>ChatGPT</source><year>2025</year><access-date>2026-07-16</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://chatgpt.com">https://chatgpt.com</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Clusmann</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kolbinger</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Muti</surname><given-names>HS</given-names> </name><etal/></person-group><article-title>The future landscape of large language models in medicine</article-title><source>Commun Med (Lond)</source><year>2023</year><month>10</month><day>10</day><volume>3</volume><issue>1</issue><fpage>141</fpage><pub-id pub-id-type="doi">10.1038/s43856-023-00370-1</pub-id><pub-id pub-id-type="medline">37816837</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Truhn</surname><given-names>D</given-names> </name><name name-style="western"><surname>Reis-Filho</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Kather</surname><given-names>JN</given-names> </name></person-group><article-title>Large language models should be used as scientific reasoning engines, not knowledge databases</article-title><source>Nat Med</source><year>2023</year><month>12</month><volume>29</volume><issue>12</issue><fpage>2983</fpage><lpage>2984</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02594-z</pub-id><pub-id pub-id-type="medline">37853138</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hassanein</surname><given-names>FEA</given-names> </name><name name-style="western"><surname>Hussein</surname><given-names>RR</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>Y</given-names> </name><name name-style="western"><surname>El-Guindy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Abou-Bakr</surname><given-names>A</given-names> </name></person-group><article-title>Calibration of AI large language models with human subject matter experts for grading of clinical short-answer responses in dental education</article-title><source>BMC Oral Health</source><year>2026</year><month>02</month><day>6</day><volume>26</volume><issue>1</issue><fpage>286</fpage><pub-id pub-id-type="doi">10.1186/s12903-026-07665-4</pub-id><pub-id pub-id-type="medline">41652418</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Namaziandost</surname><given-names>E</given-names> </name><name name-style="western"><surname>&#x00C7;elik</surname><given-names>F</given-names> </name><name name-style="western"><surname>Duran</surname><given-names>V</given-names> </name></person-group><article-title>Feedback valence and framing in AI-mediated EFL learning: a quantum-inspired analysis of their effects on goal orientation, motivational affect, and task persistence through achievement goal theory</article-title><source>Learn Motiv</source><year>2025</year><month>11</month><volume>92</volume><fpage>102200</fpage><pub-id pub-id-type="doi">10.1016/j.lmot.2025.102200</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Langren&#x00E9;</surname><given-names>N</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>S</given-names> </name></person-group><article-title>Unleashing the potential of prompt engineering for large language models</article-title><source>Patterns (N Y)</source><year>2025</year><volume>6</volume><issue>6</issue><fpage>101260</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2025.101260</pub-id><pub-id pub-id-type="medline">40575123</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Knoth</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tolzin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Janson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Leimeister</surname><given-names>JM</given-names> </name></person-group><article-title>AI literacy and its implications for prompt engineering strategies</article-title><source>Comput Educ Artif Intell</source><year>2024</year><month>06</month><volume>6</volume><fpage>100225</fpage><pub-id pub-id-type="doi">10.1016/j.caeai.2024.100225</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hassanein</surname><given-names>FEA</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Maher</surname><given-names>S</given-names> </name><name name-style="western"><surname>Barbary</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Abou-Bakr</surname><given-names>A</given-names> </name></person-group><article-title>Prompt-dependent performance of multimodal AI model in oral diagnosis: a comprehensive analysis of accuracy, narrative quality, calibration, and latency versus human experts</article-title><source>Sci Rep</source><year>2025</year><month>10</month><day>30</day><volume>15</volume><issue>1</issue><fpage>37932</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-22979-z</pub-id><pub-id pub-id-type="medline">41168327</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>D. Kulkarni</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tupsakhare</surname><given-names>P</given-names> </name></person-group><article-title>Crafting effective prompts: enhancing AI performance through structured input design</article-title><source>J Recent Trends Comput Sci Eng</source><year>2024</year><volume>12</volume><issue>5</issue><fpage>1</fpage><lpage>10</lpage><pub-id pub-id-type="doi">10.70589/JRTCSE.2024.5.1</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jacobsen</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Weber</surname><given-names>KE</given-names> </name></person-group><article-title>The promises and pitfalls of large language models as feedback providers: a study of prompt engineering and the quality of AI-driven feedback</article-title><source>AI</source><year>2025</year><volume>6</volume><issue>2</issue><fpage>35</fpage><pub-id pub-id-type="doi">10.3390/ai6020035</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Claman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sezgin</surname><given-names>E</given-names> </name></person-group><article-title>Artificial intelligence in dental education: opportunities and challenges of large language models and multimodal foundation models</article-title><source>JMIR Med Educ</source><year>2024</year><month>09</month><day>27</day><volume>10</volume><fpage>e52346</fpage><pub-id pub-id-type="doi">10.2196/52346</pub-id><pub-id pub-id-type="medline">39331527</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Mellers</surname><given-names>BA</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Smelser</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Baltes</surname><given-names>PB</given-names> </name></person-group><article-title>Decision research: behavioral</article-title><source>International Encyclopedia of the Social &#x0026; Behavioral Sciences</source><year>2001</year><publisher-name>Pergamon</publisher-name><fpage>3318</fpage><lpage>3323</lpage><pub-id pub-id-type="doi">10.1016/B0-08-043076-7/00626-4</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meyer</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jansen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Schiller</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Using LLMs to bring evidence-based feedback into the classroom: AI-generated feedback increases secondary students&#x2019; text revision, motivation, and positive emotions</article-title><source>Comput Educ Artif Intell</source><year>2024</year><month>06</month><volume>6</volume><fpage>100199</fpage><pub-id pub-id-type="doi">10.1016/j.caeai.2023.100199</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Othman</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Sharqawi</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>MohammedAziz</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>WA</given-names> </name><name name-style="western"><surname>Alatiyyah</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Mirah</surname><given-names>MA</given-names> </name></person-group><article-title>Assessing the accuracy and completeness of AI-generated dental responses: an evaluation of the Chat-GPT model</article-title><source>Healthcare (Don Mills)</source><year>2025</year><volume>13</volume><issue>17</issue><fpage>2144</fpage><pub-id pub-id-type="doi">10.3390/healthcare13172144</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Palcanis</surname><given-names>KG</given-names> </name><name name-style="western"><surname>Geiger</surname><given-names>BF</given-names> </name><name name-style="western"><surname>O&#x2019;Neal</surname><given-names>MR</given-names> </name><etal/></person-group><article-title>Preparing students to practice evidence-based dentistry: a mixed methods conceptual framework for curriculum enhancement</article-title><source>J Dent Educ</source><year>2012</year><month>12</month><volume>76</volume><issue>12</issue><fpage>1600</fpage><lpage>1614</lpage><pub-id pub-id-type="doi">10.1002/j.0022-0337.2012.76.12.tb05423.x</pub-id><pub-id pub-id-type="medline">23225679</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alkattan</surname><given-names>R</given-names> </name><name name-style="western"><surname>Alreshaid</surname><given-names>L</given-names> </name></person-group><article-title>The effectiveness of live and prerecorded video demonstrations in teaching restorative dentistry to undergraduate students: cohort study</article-title><source>JMIR Form Res</source><year>2025</year><month>09</month><day>25</day><volume>9</volume><fpage>e74383</fpage><pub-id pub-id-type="doi">10.2196/74383</pub-id><pub-id pub-id-type="medline">40997325</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>&#x00C7;akan</surname><given-names>KN</given-names> </name><name name-style="western"><surname>&#x0130;pek</surname><given-names>&#x0130;</given-names> </name></person-group><article-title>From lecture hall to clinic: dental students&#x2019; AI readiness and anxiety across educational stages</article-title><source>BMC Med Educ</source><year>2025</year><month>11</month><day>11</day><volume>25</volume><issue>1</issue><fpage>1577</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-08181-9</pub-id><pub-id pub-id-type="medline">41219925</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kong</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fok</surname><given-names>EHW</given-names> </name><name name-style="western"><surname>Yiu</surname><given-names>CKY</given-names> </name></person-group><article-title>A scoping review of large language models in dental education: applications, challenges, and prospects</article-title><source>Int Dent J</source><year>2025</year><month>12</month><volume>75</volume><issue>6</issue><fpage>103854</fpage><pub-id pub-id-type="doi">10.1016/j.identj.2025.103854</pub-id><pub-id pub-id-type="medline">40945315</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abd-Alrazaq</surname><given-names>A</given-names> </name><name name-style="western"><surname>AlSaad</surname><given-names>R</given-names> </name><name name-style="western"><surname>Alhuwail</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Large language models in medical education: opportunities, challenges, and future directions</article-title><source>JMIR Med Educ</source><year>2023</year><month>06</month><day>1</day><volume>9</volume><fpage>e48291</fpage><pub-id pub-id-type="doi">10.2196/48291</pub-id><pub-id pub-id-type="medline">37261894</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Correia</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Hickey</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>F</given-names> </name></person-group><article-title>Realizing the possibilities of the large language models: strategies for prompt engineering in educational inquiries</article-title><source>Theory Pract</source><year>2025</year><volume>64</volume><issue>4</issue><fpage>434</fpage><lpage>447</lpage><pub-id pub-id-type="doi">10.1080/00405841.2025.2528545</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>B&#x00F6;hme Kristensen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Asimakopoulou</surname><given-names>K</given-names> </name><name name-style="western"><surname>Scambler</surname><given-names>S</given-names> </name></person-group><article-title>Enhancing patient-centred care in dentistry: a narrative review</article-title><source>Br Med Bull</source><year>2023</year><month>12</month><day>11</day><volume>148</volume><issue>1</issue><fpage>79</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1093/bmb/ldad026</pub-id><pub-id pub-id-type="medline">37838360</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brondani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alves</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ribeiro</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Artificial intelligence, ChatGPT, and dental education: implications for reflective assignments and qualitative research</article-title><source>J Dent Educ</source><year>2024</year><month>12</month><volume>88</volume><issue>12</issue><fpage>1671</fpage><lpage>1680</lpage><pub-id pub-id-type="doi">10.1002/jdd.13663</pub-id><pub-id pub-id-type="medline">38973069</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shusterman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Waters</surname><given-names>AC</given-names> </name><name name-style="western"><surname>O&#x2019;Neill</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bangs</surname><given-names>M</given-names> </name><name name-style="western"><surname>Luu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Tucker</surname><given-names>DM</given-names> </name></person-group><article-title>An active inference strategy for prompting reliable responses from large language models in medical practice</article-title><source>NPJ Digit Med</source><year>2025</year><month>02</month><day>22</day><volume>8</volume><issue>1</issue><fpage>119</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01516-2</pub-id><pub-id pub-id-type="medline">39987335</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhattacharyya</surname><given-names>M</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Bhattacharyya</surname><given-names>D</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>LE</given-names> </name></person-group><article-title>High rates of fabricated and inaccurate references in ChatGPT-generated medical content</article-title><source>Cureus</source><year>2023</year><month>05</month><volume>15</volume><issue>5</issue><fpage>e39238</fpage><pub-id pub-id-type="doi">10.7759/cureus.39238</pub-id><pub-id pub-id-type="medline">37337480</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Walters</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Wilder</surname><given-names>EI</given-names> </name></person-group><article-title>Fabrication and errors in the bibliographic citations generated by ChatGPT</article-title><source>Sci Rep</source><year>2023</year><month>09</month><day>7</day><volume>13</volume><issue>1</issue><fpage>14045</fpage><pub-id pub-id-type="doi">10.1038/s41598-023-41032-5</pub-id><pub-id pub-id-type="medline">37679503</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>L</given-names> </name><name name-style="western"><surname>Schellaert</surname><given-names>W</given-names> </name><name name-style="western"><surname>Mart&#x00ED;nez-Plumed</surname><given-names>F</given-names> </name><name name-style="western"><surname>Moros-Daval</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ferri</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hern&#x00E1;ndez-Orallo</surname><given-names>J</given-names> </name></person-group><article-title>Larger and more instructable language models become less reliable</article-title><source>Nature</source><year>2024</year><month>10</month><volume>634</volume><issue>8032</issue><fpage>61</fpage><lpage>68</lpage><pub-id pub-id-type="doi">10.1038/s41586-024-07930-y</pub-id><pub-id pub-id-type="medline">39322679</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gravel</surname><given-names>J</given-names> </name><name name-style="western"><surname>D&#x2019;Amours-Gravel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Osmanlliu</surname><given-names>E</given-names> </name></person-group><article-title>Learning to fake it: limited responses and fabricated references provided by ChatGPT for medical questions</article-title><source>Mayo Clin Proc Digit Health</source><year>2023</year><volume>1</volume><issue>3</issue><fpage>226</fpage><lpage>234</lpage><pub-id pub-id-type="doi">10.1016/j.mcpdig.2023.05.004</pub-id><pub-id pub-id-type="medline">40206627</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McGowan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gui</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Dobbs</surname><given-names>M</given-names> </name><etal/></person-group><article-title>ChatGPT and Bard exhibit spontaneous citation fabrication during psychiatry literature search</article-title><source>Psychiatry Res</source><year>2023</year><month>08</month><volume>326</volume><fpage>115334</fpage><pub-id pub-id-type="doi">10.1016/j.psychres.2023.115334</pub-id><pub-id pub-id-type="medline">37499282</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompts by framing and evidence requirement.</p><media xlink:href="mededu_v12i1e90736_app1.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Coding schema.</p><media xlink:href="mededu_v12i1e90736_app2.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Sample prompts and GPT-5 responses.</p><media xlink:href="mededu_v12i1e90736_app3.pdf" xlink:title="PDF File, 262 KB"/></supplementary-material></app-group></back></article>