<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Educ</journal-id><journal-id journal-id-type="publisher-id">mededu</journal-id><journal-id journal-id-type="index">20</journal-id><journal-title>JMIR Medical Education</journal-title><abbrev-journal-title>JMIR Med Educ</abbrev-journal-title><issn pub-type="epub">2369-3762</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v12i1e84266</article-id><article-id pub-id-type="doi">10.2196/84266</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Performance of Cloud-Hosted Large Vision-Language Models on the Japanese National Examination for Clinical Laboratory Technicians: Comparative Benchmarking Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Murakami</surname><given-names>Kota</given-names></name><degrees>MMSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Matsuzawa</surname><given-names>Ryosuke</given-names></name><degrees>MMSc</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Tahara-Arai</surname><given-names>Yuya</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Ozaki</surname><given-names>Haruka</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>Master&#x2019;s Program in Medical Sciences, Graduate School of Comprehensive Human Sciences, University of Tsukuba</institution><addr-line>Ibaraki</addr-line><country>Japan</country></aff><aff id="aff2"><institution>Bioinformatics Laboratory, Institute of Medicine, University of Tsukuba</institution><addr-line>Ibaraki</addr-line><country>Japan</country></aff><aff id="aff3"><institution>Laboratory for AI Biology, RIKEN Center for Biosystems Dynamics Research</institution><addr-line>6-7-1 Minatojima Minamimachi, Chuo-ku, Kobe</addr-line><addr-line>Hyogo</addr-line><country>Japan</country></aff><aff id="aff4"><institution>Doctoral Program in Medical Sciences, Graduate School of Comprehensive Human Sciences, University of Tsukuba</institution><addr-line>Ibaraki</addr-line><country>Japan</country></aff><aff id="aff5"><institution>Ph.D. Program in Humanics, School of Integrative and Global Majors, University of Tsukuba</institution><addr-line>Ibaraki</addr-line><country>Japan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Brini</surname><given-names>Stefano</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Gad</surname><given-names>Ahmed G</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Ishida</surname><given-names>Kai</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Haruka Ozaki, PhD, Laboratory for AI Biology, RIKEN Center for Biosystems Dynamics Research, 6-7-1 Minatojima Minamimachi, Chuo-ku, Kobe, Hyogo, 650-0047, Japan, 81 78-306-0111; <email>ai-biology@ml.riken.jp</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>11</day><month>9</month><year>2026</year></pub-date><volume>12</volume><elocation-id>e84266</elocation-id><history><date date-type="received"><day>17</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>22</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>23</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Kota Murakami, Ryosuke Matsuzawa, Yuya Tahara-Arai, Haruka Ozaki. Originally published in JMIR Medical Education (<ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org">https://mededu.jmir.org</ext-link>), 11.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Education, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mededu.jmir.org/">https://mededu.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mededu.jmir.org/2026/1/e84266"/><abstract><sec><title>Background</title><p>Cloud-hosted large vision-language models (LVLMs) often outperform open-weight models on multimodal benchmarks, but their applicability to health care examinations that require both text and image reasoning remains unclear. Existing studies on Japan&#x2019;s National Examination for Clinical Laboratory Technicians mainly used text-only settings and a limited set of models.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the accuracy and applicability of cloud-hosted LVLMs on this examination, quantify the contribution of images, examine alignment with human performance across medical subfields, and test generalizability on questions administered after the models&#x2019; knowledge cutoff dates. We aim to provide a standardized benchmark of baseline multimodal performance under deployment-realistic conditions without additional fine-tuning.</p></sec><sec sec-type="methods"><title>Methods</title><p>For this comparative benchmarking study, we built the Kensagishi question-answer dataset (Kensagishi QA), a benchmark of 800 questions from the examination (2020&#x2010;2023), including text-only and image-based questions, in a standardized structured JSON format, and strict scoring criteria (all correct options required). We tested GPT-5, GPT-4o, GPT-4o-mini, Gemini 1.5 Pro, Gemini 2.5 Pro, and Neva-22B using a uniform &#x201C;numbers-only&#x201D; response prompt; images were provided as Base64-encoded data when available. Accuracy was computed overall, by item type, and by medical subfield. Human reference was drawn from published examination results. Spearman correlation assessed model-human alignment. Generalization was evaluated on 200 questions from the 71st examination (2025), released after the models&#x2019; knowledge cutoffs.</p></sec><sec sec-type="results"><title>Results</title><p>Top-performing models exceeded the 60% passing threshold. GPT-5 achieved 93.9% (95% CI 92.0%&#x2010;95.4%) accuracy with images and 88.2% (95% CI 85.7%&#x2010;90.2%) without images, Gemini 2.5 Pro 92.5% (95% CI 90.5%&#x2010;94.1%) and 86.8% (95% CI 84.2%&#x2010;88.9%), and GPT-4o 81.7% (95% CI 78.8%&#x2010;84.2%) and 79.8% (95% CI 76.8%&#x2010;82.4%), respectively. Providing images markedly improved accuracy for GPT-5, Gemini 2.5 Pro, and GPT-4o, whereas Neva-22B showed no improvement. Model-human rank correlations by medical subfield were moderate for GPT-4o (Spearman &#x03C1;=0.61; 95% CI &#x2013;0.03 to 0.90; <italic>P</italic>=.06 with images) and statistically significant for Gemini 1.5 Pro without images (&#x03C1;=0.67; 95% CI 0.07&#x2010;0.91; <italic>P</italic>=.03). On 200 postknowledge cutoff questions from the 71st examination (2025), model accuracies were comparable to historical averages (eg, GPT-5 91.5%, 95% CI 86.8%&#x2010;94.6%; Gemini 2.5 Pro 90.7%, 95% CI 85.8%&#x2010;94.0%; GPT-4o 82.0%, 95% CI 76.1%&#x2010;86.7%), indicating no material performance degradation on unseen questions.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Cloud-hosted LVLMs show strong performance on this examination, and visual input is a key driver of accuracy for leading models. Models that better mirror human difficulty patterns (eg, GPT-4o) offer complementary insights for educational analytics, whereas the highest-accuracy models (eg, GPT-5 and Gemini 2.5 Pro) may be preferable when correctness is paramount. To our knowledge, this is the first systematic, multimodel evaluation isolating visual input that extends prior text-only, few-model studies and provides Kensagishi QA to guide educational deployment of medical AI.</p></sec></abstract><kwd-group><kwd>large vision-language models</kwd><kwd>Japanese National Examination for Clinical Laboratory Technicians</kwd><kwd>multimodal reasoning</kwd><kwd>medical AI</kwd><kwd>model performance evaluation</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Large language models (LLMs) are a type of natural language processing technology that acquire the ability to understand and generate human language by learning from massive amounts of textual data. These models have demonstrated high performance across a wide range of tasks, including question answering, summarization, translation, and reasoning, with rapidly expanding applications [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. In recent years, large vision-language models (LVLMs), which are capable of handling not only text but also multimodal information including images, have also emerged. These models are designed to understand and generate both visual and textual information simultaneously, enabling their application to a broader range of tasks such as image-based question answering, image captioning, and visual reasoning [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. Notably, recent multimodal benchmarks (eg, VisionArena [<xref ref-type="bibr" rid="ref5">5</xref>] and MMStar [<xref ref-type="bibr" rid="ref6">6</xref>]) indicate that proprietary, cloud-hosted, closed-weight LVLMs currently outperform open-weight, self-hosted models.</p><p>The rapid advancement of LLMs has also extended into the medical domain, where evaluations of performance on the United States Medical Licensing Examination (USMLE) have become increasingly common. Prior studies using benchmark datasets such as MedQA, which contains USMLE-style questions, have demonstrated that LLMs can answer questions requiring domain-specific knowledge [<xref ref-type="bibr" rid="ref7">7</xref>]. For instance, Med-PaLM 2, developed by Google, achieved an accuracy of 86.5% on the MedQA dataset and received favorable evaluations compared with human physicians&#x2019; responses [<xref ref-type="bibr" rid="ref8">8</xref>]. In addition, the LLM-MedQA system using the Llama3.1:70B model reported approximately 7% improvements in accuracy and <italic>F</italic><sub>1</sub>-scores under zero-shot settings by incorporating a multiagent architecture and similar case-generation techniques [<xref ref-type="bibr" rid="ref9">9</xref>]. Furthermore, with the integration of Semantic Clinical AI (SCAI), which embeds structured clinical knowledge into LLMs, performance has significantly improved, achieving up to 95.1% accuracy on USMLE Step 3, surpassing conventional models by a considerable margin [<xref ref-type="bibr" rid="ref10">10</xref>]. In a similar effort in the Japanese context, a benchmark dataset named IgakuQA, based on questions from the Japanese National Medical Licensing Examination, was developed to explore the applicability of LLMs while accounting for linguistic and health care system differences [<xref ref-type="bibr" rid="ref11">11</xref>]. Moreover, LLMs have also been evaluated in supporting the creation of multiple-choice questions for the Japanese National Nursing Examination, particularly in generating and assessing distractors [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>In contrast, although text-only LLMs have been actively evaluated on medical licensing examinations, the applicability of LVLMs to health care&#x2013;related national examinations&#x2014;particularly those requiring both textual and visual reasoning&#x2014;remains underexplored. For the Japanese National Examination for Clinical Laboratory Technicians, existing reports have primarily assessed a limited number of general-purpose LLMs, such as ChatGPT-3.5 and GPT-4 [<xref ref-type="bibr" rid="ref13">13</xref>-<xref ref-type="bibr" rid="ref15">15</xref>], often in text-only settings. As a result, it is still unclear how modern, cloud-hosted (closed-weight) LVLMs perform on this examination and to what extent visual input contributes to accuracy.</p><p>This study identifies 2 primary gaps. First, there is no publicly available, well-structured dataset specifically tailored to the National Examination for Clinical Laboratory Technicians (including paired image assets and metadata). Consequently, considerable effort is required to conduct reproducible cross-model comparisons or assess newly developed systems. Second, prior work has not provided a systematic, multimodel evaluation of LVLMs that isolates the contribution of images; although some attempts have included image-dependent items (eg, peripheral blood smear morphology classification and electrophoresis band pattern interpretation), coverage has been limited and insufficient for comprehensive multimodal comparison. Taken together, these gaps highlight the need for a standardized, reproducible benchmark that enables fair cross-model comparison under controlled conditions.</p><p>Given this background, the aim of the present study was to conduct a multifaceted evaluation of the applicability of cloud-hosted LVLMs to the medical domain, using the National Examination for Clinical Laboratory Technicians as a case study. In particular, we focused on the impact of visual information, in addition to textual information, on model accuracy. To this end, we constructed a custom benchmark dataset, Kensagishi question-answer (Kensagishi QA), based on past questions from the Japanese National Examination for Clinical Laboratory Technicians. By systematically comparing the performance of multiple cloud-hosted LVLMs, we sought to elucidate their strengths and limitations and to provide fundamental insights to support future advancements in medical AI.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Construction of a Dataset Based on the Japanese National Examination for Clinical Laboratory Technicians</title><p>This study was designed as a comparative benchmarking study to develop an examination benchmark and evaluate the performance of multiple cloud-hosted LVLMs. To support this evaluation, we constructed a novel benchmark dataset, Kensagishi QA, comprising questions from the 66th to 69th National Examination for Clinical Laboratory Technicians conducted between 2020 and 2023. The official examination questions for each year are publicly available in PDF format [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. These materials were obtained from the official website of the Ministry of Health, Labour and Welfare (MHLW), where the content is provided under the Public Data License (version 1.0; PDL1.0), unless otherwise noted. Redistribution of the structured question data in our GitHub repository complied with the applicable license terms, including appropriate attribution.</p><p>From these PDF files, we extracted the question and answer choices and converted them into a machine-readable JSON format (<xref ref-type="fig" rid="figure1">Figure 1</xref>). During the conversion process, the data were structured to include fields such as question ID, question text, answer choices (1-5), correct answer, and question type (text/image).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Example of a structured dataset entry from the Japanese National Examination for Clinical Laboratory Technicians. The question and choices were translated from the original Japanese text for illustrative purposes.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig01.png"/></fig><p>The National Examination for Clinical Laboratory Technicians includes a mix of text-only questions (hereafter referred to as text questions) and questions containing medical images or diagrams (hereafter referred to as image questions). In the constructed dataset, these questions were categorized by type, allowing for separate performance evaluations of each question format.</p><p>All questions follow a 5-choice format and consist of either single-answer or multiple-answer types, where a single question may have more than one correct option. Each year, 200 questions are administered&#x2014;100 in the morning and 100 in the afternoon&#x2014;resulting in a total of 800 questions collected from consecutive examination years (2020&#x2010;2023). Of the 800 questions, 670 (83.75%) were classified as text questions, whereas 130 (16.25%) were identified as image questions. The image questions include prompts that require interpretation of visual information, such as blood smears, histopathological images, device output screens, and tabular data, thereby necessitating visual input processing.</p></sec><sec id="s2-2"><title>Evaluation of Accuracy of LVLMs on the Japanese National Examination for Clinical Laboratory Technicians</title><p>In this study, we evaluated the accuracy of multiple LVLMs on questions from the National Examination for Clinical Laboratory Technicians. To specifically assess performance on questions involving images, we selected LVLMs capable of processing both textual and visual inputs.</p><p>The selection of these 6 LVLMs was intended to represent a range of model generations, providers, and practical deployment scenarios rather than to compare only the most recent state-of-the-art systems. In particular, we included multiple models released by the same provider (GPT-4o-mini, GPT-4o, and GPT-5) to examine performance differences across model generations and to assess whether earlier, lower-cost models remain viable for large-scale or educational use, in addition to evaluating the capabilities of the latest model. This reflects realistic constraints in applied settings, where the most recent model is not always adopted due to factors such as inference cost, availability, latency, and system compatibility. Furthermore, models from different providers (OpenAI, Google, and NVIDIA) were included to reduce vendor-specific bias and examine how differences in architecture and training affect multimodal performance on the same benchmark. We note that all evaluated models were accessed via cloud APIs in this study, and no local deployment or on-premises inference was performed.</p><p>An overview of the models used in this study is presented in <xref ref-type="table" rid="table1">Table 1</xref>. For Gemini 2.5 Pro, we used the preview version released on March 25, 2025 (gemini-2.5-pro-preview-03-25), prior to its general availability.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Comparison of providers, API names, parameter sizes, release dates, and knowledge cutoffs for each large vision-language model.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Provider</td><td align="left" valign="bottom">API name</td><td align="left" valign="bottom">Number of parameters</td><td align="left" valign="bottom">Release date</td><td align="left" valign="bottom">Knowledge cutoff</td></tr></thead><tbody><tr><td align="left" valign="top">GPT-4o-mini</td><td align="left" valign="top">OpenAI</td><td align="left" valign="top">gpt-4o-mini-2024-07-18</td><td align="left" valign="top">Not disclosed</td><td align="left" valign="top">July 18, 2024</td><td align="left" valign="top">October 2023</td></tr><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">OpenAI</td><td align="left" valign="top">gpt-4o-2024-08-06</td><td align="left" valign="top">Not disclosed</td><td align="left" valign="top">August 6, 2024</td><td align="left" valign="top">October 2023</td></tr><tr><td align="left" valign="top">Gemini 1.5 Pro</td><td align="left" valign="top">Google</td><td align="left" valign="top">gemini-1.5-pro</td><td align="left" valign="top">Not disclosed</td><td align="left" valign="top">September 24, 2024</td><td align="left" valign="top">May 2024</td></tr><tr><td align="left" valign="top">Gemini 2.5 Pro (Preview)</td><td align="left" valign="top">Google</td><td align="left" valign="top">gemini-2.5-pro-preview-03-25</td><td align="left" valign="top">Not disclosed</td><td align="left" valign="top">March 25, 2025 (Preview)</td><td align="left" valign="top">January 2025</td></tr><tr><td align="left" valign="top">Neva-22B</td><td align="left" valign="top">NVIDIA</td><td align="left" valign="top">nvidia/neva-22b</td><td align="left" valign="top">22 billion</td><td align="left" valign="top">Not disclosed</td><td align="left" valign="top">Not disclosed</td></tr><tr><td align="left" valign="top">GPT-5</td><td align="left" valign="top">OpenAI</td><td align="left" valign="top">gpt-5-2025-08-07</td><td align="left" valign="top">Not disclosed</td><td align="left" valign="top">August 7, 2025</td><td align="left" valign="top">September 30, 2024</td></tr></tbody></table></table-wrap><p>The National Examination for Clinical Laboratory Technicians includes multiple-answer questions in which more than one option may be correct. To accommodate this format, we adopted a strict evaluation criterion: a response was considered correct only if all correct answer choices were selected. Partial matches were regarded as incorrect.</p><p>In addition, some examination questions include images but can still be answered correctly based solely on the answer choices or surrounding textual context. However, from the perspective of evaluating AI model performance, it is essential to assess the impact of image presence on accuracy appropriately. Therefore, we clearly distinguished between text-based and image-based questions and calculated the accuracy for each category separately.</p><p>For image questions, the image file corresponding to each question ID was encoded in Base64 and provided to the model along with the associated question text. This enabled multimodal input handling for questions that include visual content. Instructions to each model were given based on a standardized prompt format. Specifically, the models were presented with the question text and answer choices, followed by an explicit instruction to &#x201C;respond with only the number(s) of the correct choice(s).&#x201D; The prompt content is as follows:</p><disp-quote><p>Please provide the number(s) of the correct option(s) for the question below in the following format. If the question requires selecting only one answer, provide a single number. If it requires selecting two answers (e.g., "Choose two"), provide two numbers. Do not include any explanations or reasoning&#x2014;just the number(s).</p><p>Question: {Question text}</p><p>Options: {List of options}</p><p>Please provide only the number(s) of the correct answer(s):</p></disp-quote><p>With this prompt format, the models were instructed to output only the option numbers (eg, &#x201C;2&#x201D; or &#x201C;14&#x201D;). The output was parsed using regular expressions (re.findall(r'\d+', text)) to extract numerical values, which were interpreted as the selected choice numbers. This approach was designed to robustly handle variations in the response format, such as &#x201C;1 4&#x201D; or &#x201C;1,4.&#x201D;</p><p>For all questions, each model received the question text and answer choices as textual input, and for image questions, the corresponding visual data were also provided. The images were encoded in Base64 format and provided alongside the text.</p><p>The images used in this study were screenshots extracted from the original examination materials. No image preprocessing such as resizing, cropping, or normalization was applied. Each image was used in its original resolution and directly encoded into Base64 format before being provided to the models.</p><p>The model outputs were recorded in JSONL format as pairs of question IDs and predicted choice numbers. The correctness of each prediction was determined based on the strict evaluation criterion described earlier.</p><p>Final accuracy was calculated as the proportion of correctly answered questions relative to the total number of questions. To analyze the impact of visual information on model performance, accuracy was computed separately for text questions and image questions. As a reference, we used the passing threshold for the national examination (60% accuracy) and assessed whether each model exceeded this criterion, thereby enabling comparative analysis across models.</p><p>To reduce the impact of stochastic variation inherent in the generation process, each model independently answered all questions 3 times under identical settings. The accuracy of each run was first computed, and the mean across the 3 runs was used as the model&#x2019;s final accuracy. Accuracy figures display the 3 per-run values as points and 95% Wilson score CIs as error bars.</p></sec><sec id="s2-3"><title>Categorization of Questions by Medical Subfield</title><p>In this study, question categorization by medical subfield was conducted based on the official examination guidelines for the Japanese National Examination for Clinical Laboratory Technicians published by the MHLW. Specifically, the question stems and answer choices were carefully examined to interpret the medical theme and underlying intent of each question. Based on this analysis, we manually classified each question into relevant subfields, such as clinical physiology, clinical chemistry, and pathological histocytology. The categorization was performed by a single author (KM); no independent reannotation or interannotator agreement assessment was conducted, which is acknowledged as a study limitation. Using this categorization, accuracy was calculated for each subfield.</p></sec><sec id="s2-4"><title>Source of Human Accuracy Data</title><p>The human accuracy rates used in this study were based on published results from a performance analysis of the national examination [<xref ref-type="bibr" rid="ref20">20</xref>]. Specifically, we extracted the average accuracy for each question from the previous study [<xref ref-type="bibr" rid="ref20">20</xref>], titled &#x201C;Item-by-item Accuracy Analysis,&#x201D; which presents data from the 68th National Examination for Clinical Laboratory Technicians based on responses from 1896 examinees at 40 institutions.</p></sec><sec id="s2-5"><title>Correlation Analysis Between Model and Human Accuracy</title><p>To examine the correlation between model outputs and human accuracy, we calculated the Spearman rank correlation coefficient. For each medical subfield, we computed the accuracy rates for both humans and models, ranked them in descending order, and then merged the results based on the corresponding subfield names to obtain the rank difference <italic>d<sub>i</sub></italic> for each subfield. The Spearman rank correlation coefficient, &#x03C1;<italic>,</italic> is defined by the following formula:</p><disp-formula id="equWL1"><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>&#x03C1;</mml:mi><mml:mtext>=</mml:mtext><mml:mn>1</mml:mn><mml:mtext>&#x2212;</mml:mtext><mml:mfrac><mml:mrow><mml:mn>6</mml:mn><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mtext>&#x2212;</mml:mtext><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <italic>d<sub>i</sub></italic> is the difference in rank between the human and model accuracies for subfield <italic>i</italic>, and n is the total number of subfields. This analysis quantitatively evaluates the extent to which the model&#x2019;s accuracy patterns across subfields align with those observed in human performance. All <italic>P</italic> values are reported as exact values and formatted according to established medical journal guidelines, using an italicized capital <italic>P</italic> without a leading zero; values less than .001 are reported as <italic>P</italic>&#x003C;.001. All statistical analyses were performed in Python (version 3.10.12; Python Software Foundation) using SciPy (scipy.stats, version 1.15.3; SciPy community); reported <italic>P</italic> values are two-sided, and statistical significance was defined as <italic>P</italic>&#x003C;.05.</p></sec><sec id="s2-6"><title>Statistical Comparison Between Training Years and the 2025 Examination</title><p>In addition to descriptive comparisons, we conducted a statistical significance test to examine whether the declines in accuracy between the training years (2020&#x2010;2023) and the 2025 examination were statistically meaningful. For each model and image condition, we used the mean accuracy across the 3 independent runs as the final accuracy (see above) and constructed a 2&#x00D7;2 contingency table of correct versus incorrect responses using question-level counts (n=800 for 2020&#x2010;2023 and n=200 for 2025). Because the 3 runs are repeated measurements on the same questions rather than independent samples, the denominator was set to the number of questions rather than the number of question-run pairs. A chi-square test of independence (Pearson, without continuity correction) was then applied to each comparison.</p></sec><sec id="s2-7"><title>Specifications of the Tested LVLMs</title><p><xref ref-type="table" rid="table1">Table 1</xref> summarizes the specifications of the LVLMs used in this study. For each model, we document the provider, API name used, number of parameters, release date, and the final knowledge cutoff date of the training data. All inference was performed via the providers&#x2019; cloud APIs between July 2024 and October 2025; no local or on-premises graphics processing unit (GPU) compute was used, and total inference cost was modest given the limited number of API calls (approximately 800 questions&#x00D7;6 models&#x00D7;2 image conditions). <xref ref-type="table" rid="table2">Table 2</xref> summarizes the generation settings used for each model.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Generation settings used for each model in this study.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Provider</td><td align="left" valign="bottom">Temperature</td><td align="left" valign="bottom">Top-p</td><td align="left" valign="bottom">Top-k</td><td align="left" valign="bottom">Number of candidates (n)</td></tr></thead><tbody><tr><td align="left" valign="top">GPT-4o-mini</td><td align="left" valign="top">OpenAI</td><td align="left" valign="top">0.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">Not configurable</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">OpenAI</td><td align="left" valign="top">0.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">Not configurable</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">Gemini 1.5 Pro</td><td align="left" valign="top">Google</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.95</td><td align="left" valign="top">64 (fixed)</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">Gemini 2.5 Pro (Preview)</td><td align="left" valign="top">Google</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.95</td><td align="left" valign="top">64 (fixed)</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">Neva-22B</td><td align="left" valign="top">NVIDIA</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.7</td><td align="left" valign="top">Not configurable</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">GPT-5</td><td align="left" valign="top">OpenAI</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">Not configurable</td><td align="left" valign="top">1</td></tr></tbody></table></table-wrap></sec><sec id="s2-8"><title>Calculation of CIs</title><p>For every point estimate, we reported a two-sided 95% CI. For accuracy (a binomial proportion), we used the Wilson score interval [<xref ref-type="bibr" rid="ref21">21</xref>], with the mean number of correct answers across the 3 runs as the numerator and the number of questions as the denominator (not the number of question-run pairs, because the runs were repeated measurements of the same items). The Wilson CIs reported here convey the uncertainty of each model&#x2019;s accuracy estimate; interval overlap should not be interpreted as a formal between-model comparison, and no correction for multiple comparisons was applied. For the Spearman correlation between model and human subfield accuracy, we used the Fisher <italic>z</italic> transformation with the number of subfields as the sample size (n=10). For differences in accuracy between the 2025 and the 2020&#x2010;2023 examinations and visual-heavy versus text-heavy subfields, we used the Newcombe hybrid score method [<xref ref-type="bibr" rid="ref22">22</xref>]. Because the benchmark enumerated all questions from the target examinations, these CIs express the precision of each estimate when the fixed question set is treated as a sample of underlying model competence rather than as sampling error from a larger population.</p></sec><sec id="s2-9"><title>Data Availability</title><p>The data used in this study are publicly available in a GitHub repository [<xref ref-type="bibr" rid="ref23">23</xref>]. The repository contains past question data from the National Examination for Clinical Laboratory Technicians in JSONL format.</p></sec><sec id="s2-10"><title>Evaluation of Model Generalization Using Questions After the Knowledge Cutoff</title><p>To reduce the possibility that the models derived correct answers solely from memorized training data, we conducted an additional performance evaluation using questions published after each model&#x2019;s knowledge cutoff. Specifically, we used all 200 questions (100 in the morning and 100 in the afternoon) from the 71st National Examination for Clinical Laboratory Technicians, administered in 2025 [<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>In this evaluation, the models were prompted using the same format as in the previous experiments. The generated responses were compared against the correct answers. As a reference, we calculated the average accuracy over the 4 earlier years (2020&#x2010;2023) and denoted this value as the &#x201C;historical average.&#x201D; The deviation between this historical average and the 2025 accuracy was then computed for each model. Through this evaluation, we examined each model&#x2019;s ability to generalize to previously unseen questions and assessed whether the models could answer them without relying on prior exposure to the question content.</p></sec><sec id="s2-11"><title>Ethical Considerations</title><p>This study did not involve newly conducted research on human participants. We used only (1) publicly available questions from the Japanese National Examination for Clinical Laboratory Technicians and (2) aggregated item-level mean accuracy values previously published in a report [<xref ref-type="bibr" rid="ref20">20</xref>]. The human accuracy data used in this study, namely the aggregated, item-level mean accuracy values previously published in that report, were group-level averages from 1896 examinees at 40 institutions and contained no information that could identify any individual. No individual-level data were collected or accessed, and no identifiable images are included in this manuscript or its supplementary materials. According to the Ethical Guidelines for Medical and Biological Research Involving Human Subjects, jointly issued by the Ministry of Education, Culture, Sports, Science and Technology, the MHLW, and the Ministry of Economy, Trade and Industry, the guidelines do not apply to research using only existing information that does not relate to an individual [<xref ref-type="bibr" rid="ref25">25</xref>]. Therefore, ethics review and approval and informed consent were not required for this study.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overall Accuracy of LVLMs</title><p><xref ref-type="fig" rid="figure2">Figure 2</xref> illustrates the accuracy achieved by each LVLM on all 800 questions from the National Examination for Clinical Laboratory Technicians. In this study, for image-based questions, we evaluated model performance under 2 conditions: an image-provided (image given) condition and a no-image (no image given) condition. For questions without associated images, identical inputs were used for both conditions. As evaluation criteria, we adopted the human passing threshold of 60% accuracy as well as the average human accuracy of 69.7%.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Accuracy of large vision-language models on the National Examination for Clinical Laboratory Technicians. The bar chart shows accuracies (0%&#x2010;100%) on 800 questions (text- and image-based) from the 66th to 69th examinations. The red dashed line indicates the human passing threshold of 60%, and the green dashed line represents the average human accuracy of 69.7%. Points denote the 3 per-run accuracies; error bars are 95% Wilson score CIs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig02.png"/></fig><p>The overall average accuracies showed that GPT-5 achieved 93.9% with image input and 88.2% without, both substantially exceeding the passing threshold. Gemini 2.5 Pro followed with 92.5% with images and 86.8% without, also well above the benchmark. GPT-4o surpassed the threshold as well, achieving 81.7% with images and 79.8% without. Gemini 1.5 Pro recorded 69.4% with images and 68.5% without, both above the 60% benchmark.</p><p>In contrast, GPT-4o-mini scored 56.0% with image input and 53.0% without, falling below the passing threshold under both conditions. Neva-22B showed the lowest accuracy among all models, with 29.6% (with images) and 32.1% (without images). When compared with the average human accuracy (69.7%), both Gemini 2.5 Pro and GPT-4o outperformed human examinees, regardless of image availability.</p></sec><sec id="s3-2"><title>Effect of Image Input on Model Accuracy for Image-Based Questions</title><p><xref ref-type="fig" rid="figure3">Figures 3</xref> and <xref ref-type="fig" rid="figure4">4</xref> present a comparison of answer accuracies for LVLMs on all questions (text- and image-based) and on image-based questions, respectively, under 2 conditions: image-provided (image given) and no-image (no image given). In this section, we describe in detail how the presence of visual information influences model performance across different models and sessions.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Accuracy of GPT-5, GPT-4o, GPT-4o-mini, Neva-22B, Gemini 1.5 Pro, and Gemini 2.5 Pro on all questions (text- and image-based) from the 66th to 69th National Examinations for Clinical Laboratory Technicians. The x-axis represents the examination (66th to 69th), each pooling the morning and afternoon sessions, and the y-axis represents accuracy. Blue bars indicate the condition in which images were provided for image-based questions, while orange bars indicate the condition without image input. The red dashed line marks the passing threshold of 60%. This allows comparison of the effect of image input on overall accuracy across models and examination sessions. Points denote the 3 per-run accuracies; error bars are 95% Wilson score CIs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig03.png"/></fig><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Accuracy of GPT-5, GPT-4o, GPT-4o-mini, Neva-22B, Gemini 1.5 Pro, and Gemini 2.5 Pro on image-based questions from the 66th to 69th National Examinations for Clinical Laboratory Technicians. The x-axis indicates the examination (66th to 69th), pooling the morning and afternoon sessions, and the y-axis represents accuracy. Blue bars denote the image-provided condition (image given), and orange bars represent the no-image condition (no image given). The red dashed line indicates the passing threshold of 60%. This figure enables comparison not only of performance differences among models but also of the impact of image availability on accuracy for image-based questions. Points denote the 3 per-run accuracies; error bars are 95% Wilson score CIs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig04.png"/></fig><p>On all questions, GPT-5 consistently outperformed the other models, achieving around 90.0% accuracy across all sessions, with the image-provided condition always surpassing the no-image condition. In particular, in the afternoon session of the 68th examination, GPT-5 achieved 97.0% with images and 84.7% without images, representing the highest performance among all models.</p><p>GPT-4o generally performed better with images, with accuracies ranging from 75.0% to 87.0% with images and 74.0% to 86.0% without images. Although some exceptions were observed, overall, image input improved its performance. GPT-4o-mini exhibited lower overall accuracy compared with other models, but in many sessions, performance was higher under the image-provided condition (46.3%&#x2010;60.7%) than under the no-image condition (45.3%&#x2010;57.3%).</p><p>Gemini 2.5 Pro also showed consistently higher accuracy under the image-provided condition in every session, such as 93.0% with images and 88.0% without images in the morning session of the 67th examination. Gemini 1.5 Pro showed accuracies ranging from 64.0% to 75.7% with images and 61.3% to 77.0% without. While in some sessions the no-image condition outperformed the with-image one, the overall differences were not substantial.</p><p>Neva-22B demonstrated the opposite trend, achieving accuracies of 25.0%&#x2010;37.0% with images and 28.0%&#x2010;39.0% without, indicating better performance without visual information. Moreover, frequent formatting errors were observed, such as blank responses or outputs that did not conform to the specified format.</p><p>In addition to the session-wise comparisons, we quantified the overall effect of image input using all image-based questions in Kensagishi QA (n=130), pooled across the 66th to 69th examinations (<xref ref-type="fig" rid="figure5">Figure 5</xref>). For each model, we compared accuracy under the image-provided condition (image given) against the no-image condition (no image given) and computed the absolute difference (&#x0394;Acc=image given&#x2212;no image given). Across the top-performing models, image input consistently improved accuracy: GPT-5 achieved 86.15% with images versus 52.05% without images (&#x0394;Acc=34.10 points), and Gemini 2.5 Pro achieved 79.23% versus 49.23% (&#x0394;Acc=30.00 points). GPT-4o also benefited from image input, with 56.92% versus 46.15% (&#x0394;Acc=10.77 points). GPT-4o-mini improved from 24.10% to 42.56% (&#x0394;Acc=18.46 points), while Gemini 1.5 Pro showed an increase from 37.18% to 40.51% (&#x0394;Acc=3.33 points). In contrast, Neva-22B exhibited a negative image effect, decreasing from 30.77% (no image given) to 15.13% (image given; &#x0394;Acc=&#x2013;15.64 points). These results indicate that the impact of image input varies substantially across models.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Comparison of average accuracy on image-based questions (n=130) for 6 LVLMs under the image-provided (image given) and no-image (no image given) conditions, pooled across the 66th to 69th National Examinations for Clinical Laboratory Technicians. The x-axis indicates the model, and the y-axis represents accuracy. Blue bars denote the image-given condition, and orange bars represent the no-image-given condition. Image input improved accuracy for all models except Neva-22B. The red dashed line represents the passing threshold of 60%. Points denote the 3 per-run accuracies; error bars are 95% Wilson score CIs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig05.png"/></fig></sec><sec id="s3-3"><title>Modality-Specific Analysis of Image-Based Questions</title><p>To further investigate the sources of error in image-based questions, we categorized all 130 image-based questions into 5 visual modalities: (1) photographs (eg, histology images and instrument photographs; n=91), (2) graphs (n=21), (3) schematic diagrams (n=12), (4) tables (n=5), and (5) combined table+photograph items (n=1). As shown in <xref ref-type="table" rid="table3">Table 3</xref>, clear modality-dependent differences were observed. Photograph-based questions showed substantial performance gaps across models. For example, GPT-5 achieved 84.62% accuracy on photograph-based questions, Gemini 2.5 Pro 80.22%, and GPT-4o 59.34%, whereas Neva-22B achieved only 17.95%. In contrast, schematic diagrams yielded relatively high accuracy for top-performing models (Gemini 2.5 Pro: 91.67% and GPT-5: 97.22%), suggesting that structured visual abstractions were more robustly interpreted than naturalistic medical photographs. Graph-based questions were handled well by GPT-5 (82.54%), but showed lower accuracy for Gemini 2.5 Pro (61.90%) and substantial degradation in lower-performing models. Table-based questions demonstrated large intermodel variance, with Gemini 2.5 Pro and GPT-5 achieving 100.00% accuracy, and Gemini 1.5 Pro and Neva-22B achieving 20.00%.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Accuracy by model and image modality, presented as percentages. For each model and image condition, accuracy is the mean across 3 independent runs (see Methods). The 130 image-based questions were grouped into 5 visual modalities; the n given in each column header is the number of image-based questions assigned to that modality, not a number of correct answers.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Photograph (n=91), %</td><td align="left" valign="bottom">Graph (n=21), %</td><td align="left" valign="bottom">Diagram (n=12), %</td><td align="left" valign="bottom">Table (n=5), %</td><td align="left" valign="bottom">Table+photograph (n=1), %</td></tr></thead><tbody><tr><td align="left" valign="top">GPT-4o-mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>With image</td><td align="left" valign="top">47.62</td><td align="left" valign="top">23.81</td><td align="left" valign="top">41.67</td><td align="left" valign="top">40.00</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Without image</td><td align="left" valign="top">27.11</td><td align="left" valign="top">15.87</td><td align="left" valign="top">27.78</td><td align="left" valign="top">0.00</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>With image</td><td align="left" valign="top">59.34</td><td align="left" valign="top">46.03</td><td align="left" valign="top">52.78</td><td align="left" valign="top">60.00</td><td align="left" valign="top">100.00</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Without image</td><td align="left" valign="top">52.01</td><td align="left" valign="top">20.63</td><td align="left" valign="top">44.44</td><td align="left" valign="top">40.00</td><td align="left" valign="top">100.00</td></tr><tr><td align="left" valign="top">Gemini 1.5 Pro</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>With image</td><td align="left" valign="top">44.69</td><td align="left" valign="top">38.10</td><td align="left" valign="top">25.00</td><td align="left" valign="top">20.00</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Without image</td><td align="left" valign="top">43.96</td><td align="left" valign="top">20.63</td><td align="left" valign="top">16.67</td><td align="left" valign="top">40.00</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top">Gemini 2.5 Pro</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>With image</td><td align="left" valign="top">80.22</td><td align="left" valign="top">61.90</td><td align="left" valign="top">91.67</td><td align="left" valign="top">100.00</td><td align="left" valign="top">100.00</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Without image</td><td align="left" valign="top">52.75</td><td align="left" valign="top">38.10</td><td align="left" valign="top">58.33</td><td align="left" valign="top">20.00</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top">Neva-22B</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>With image</td><td align="left" valign="top">17.95</td><td align="left" valign="top">3.17</td><td align="left" valign="top">13.89</td><td align="left" valign="top">20.00</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Without image</td><td align="left" valign="top">35.16</td><td align="left" valign="top">19.05</td><td align="left" valign="top">25.00</td><td align="left" valign="top">20.00</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top">GPT-5</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>With image</td><td align="left" valign="top">84.62</td><td align="left" valign="top">82.54</td><td align="left" valign="top">97.22</td><td align="left" valign="top">100.00</td><td align="left" valign="top">100.00</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Without image</td><td align="left" valign="top">56.04</td><td align="left" valign="top">46.03</td><td align="left" valign="top">30.56</td><td align="left" valign="top">46.67</td><td align="left" valign="top">100.00</td></tr></tbody></table></table-wrap></sec><sec id="s3-4"><title>Model Accuracy Across Medical Subfields</title><p>To assess how model performance varies across different areas of medical knowledge, we aggregated the accuracy of each LVLM by medical subfield (<xref ref-type="fig" rid="figure6">Figure 6</xref>). The 10 targeted subfields were clinical chemistry, clinical hematology, clinical immunology, clinical microbiology, clinical physiology, general clinical laboratory medicine, general clinical laboratory science, introduction to medical engineering, pathological histocytology, and public health.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Accuracy by medical subfield for each model&#x2014;GPT-5, GPT-4o, GPT-4o-mini, Neva-22B, Gemini 1.5 Pro, and Gemini 2.5 Pro&#x2014;on all questions (both text- and image-based) from the 66th to 69th National Examinations for Clinical Laboratory Technicians. The x-axis represents medical subfields, and the y-axis represents accuracy. Blue bars indicate accuracy under the image-given condition (where images were provided for all questions), while orange bars represent the no-image-given condition (no images provided, including for image-based questions). The red dashed line indicates the passing threshold of 60%. For each model, points denote the 3 per-run accuracies, and error bars are 95% Wilson score CIs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig06.png"/></fig><p>GPT-5 achieved exceptionally high performance across all subfields, with accuracies ranging from approximately 90.0% to 100.0%. In the introduction to medical engineering, it reached 100.0%, while in general clinical laboratory medicine, it achieved 99.2% under the image-given condition and 97.5% under the no-image-given condition. Similarly, in clinical chemistry and clinical microbiology, GPT-5 maintained stable accuracy around 95.0% regardless of image availability. Although its performance declined slightly in the no-image-given condition for clinical physiology (80.1%) and pathological histocytology (81.3%), accuracy remained above 80.0% in all subfields.</p><p>Gemini 2.5 Pro demonstrated consistently high accuracy across most subfields, typically in the 80.0%&#x2010;90.0% range, and achieved 100.0% accuracy in both general clinical laboratory medicine and introduction to medical engineering. GPT-4o also showed robust performance, with accuracy generally in the 70.0%&#x2010;80.0% range and exceeding 90.0% in general clinical laboratory medicine.</p><p>In contrast, Gemini 1.5 Pro exhibited substantial variability across subfields. Although some sessions in general clinical laboratory medicine reached 100.0% accuracy, performance in clinical physiology and public health fell below 50.0% in several sessions. GPT-4o-mini and Neva-22B showed lower overall accuracy across subfields, with Neva-22B remaining below 50.0% accuracy in all subfields.</p><p>To contextualize differences in subfield-level performance, we quantified the proportion of image-based questions within each subfield and classified them into visual-heavy and text-heavy subfields. Clinical physiology (37/104, 35.6%), clinical hematology (24/72, 33.3%), pathological histocytology (23/112, 20.5%), clinical immunology (14/88, 15.9%), and general clinical laboratory science (12/80, 15.0%) were categorized as visual-heavy subfields. In contrast, clinical microbiology, introduction to medical engineering, public health, general clinical laboratory medicine, and clinical chemistry contained relatively few image-based questions and were categorized as text-heavy.</p><p>Grouping subfields into visual-heavy and text-heavy categories revealed a consistent trend across models (<xref ref-type="fig" rid="figure7">Figure 7</xref>). All models demonstrated lower accuracy in visual-heavy subfields compared with text-heavy subfields, as indicated by negative effect sizes (visual-heavy minus text-heavy). The magnitude of this performance gap varied by model capacity: GPT-5 and Gemini 2.5 Pro exhibited relatively small differences, indicating robustness to visually demanding content, whereas GPT-4o-mini and Neva-22B showed substantially larger negative effect sizes. These findings suggest that visually intensive disciplines pose greater challenges for lower-capacity models and that higher-capacity models are more resilient to the increased cognitive and multimodal demands of visual-heavy medical subfields.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Effect size of visual-heavy subfields under the image-given condition, defined as the difference in accuracy between visual-heavy and text-heavy subfields (visual-heavy minus text-heavy). Negative values indicate lower accuracy in visual-heavy domains. All models exhibited a performance drop in visual-heavy subfields, with larger drops observed for GPT-4o-mini and Neva-22B and smaller drops for GPT-5 and Gemini 2.5 Pro, indicating greater robustness of higher-capacity models to visually demanding content. Points denote the 3 per-run effect sizes; the error bar is the 95% Newcombe hybrid-score CI for the visual-minus-text difference.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig07.png"/></fig></sec><sec id="s3-5"><title>Correlation With Human Performance</title><p>To evaluate the consistency between model outputs and human accuracy rankings by subfield, we calculated the Spearman rank correlation coefficient (<xref ref-type="fig" rid="figure8">Figure 8</xref>). This metric indicates how closely each model mirrors the domain-specific performance trends observed in human examinees; a coefficient closer to 1 implies stronger alignment with human rankings.</p><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Comparison of Spearman rank correlation coefficients (&#x03C1;) for each model under image-provided (image given) and no-image (no image given) conditions, with exact <italic>P</italic> values reported in the main text. Based on human and model accuracy, medical subfields were ranked, and Spearman rank correlation was calculated. The red dashed line indicates a reference value for strong correlation (&#x03C1;=0.7), while the green dashed line indicates a reference value for moderate correlation (&#x03C1;=0.4). This figure quantitatively illustrates how closely each model&#x2019;s performance aligns with human subfield-level scoring trends. Points denote the Spearman coefficient recomputed from each of the 3 individual runs, and error bars are 95% Fisher z CIs (n=10 subfields).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig08.png"/></fig><p>GPT-4o showed relatively high correlations, with &#x03C1;=0.61 (95% CI &#x2013;0.03 to 0.90; <italic>P</italic>=.06) under the image-provided condition and &#x03C1;=0.52 (95% CI &#x2013;0.17 to 0.86; <italic>P</italic>=.13) under the no-image condition, though neither reached statistical significance. GPT-4o-mini exhibited similar results, achieving &#x03C1;=0.61 (95% CI &#x2013;0.03 to 0.90; <italic>P</italic>=.06) with images and &#x03C1;=0.54 (95% CI &#x2013;0.14 to 0.87; <italic>P</italic>=.11) without images, with the image-provided condition approaching significance.</p><p>GPT-5 showed &#x03C1;=0.32 (95% CI &#x2013;0.39 to 0.79; <italic>P</italic>=.37) under the image-provided condition, and &#x03C1;=0.28 (95% CI &#x2013;0.42 to 0.78; <italic>P</italic>=.43) under the no-image condition. These weaker correlations indicate limited consistency with human rankings.</p><p>Gemini 1.5 Pro recorded the highest correlation under the no-image condition, with &#x03C1;=0.67 (95% CI 0.07&#x2010;0.91; <italic>P</italic>=.03), which reached statistical significance. In contrast, under the image-provided condition it showed a lower correlation (&#x03C1;=0.48, 95% CI &#x2013;0.22 to 0.85; <italic>P</italic>=.16), not reaching significance.</p><p>Gemini 2.5 Pro showed moderate correlations&#x2014;&#x03C1;=0.45 (95% CI &#x2013;0.25 to 0.84; <italic>P</italic>=.19) without images and &#x03C1;=0.41 (95% CI &#x2013;0.30 to 0.83; <italic>P</italic>=.24) with images&#x2014;but neither was significant. Neva-22B exhibited low correlations in both conditions (&#x03C1;=0.43, 95% CI &#x2013;0.27 to 0.83; <italic>P</italic>=.21 without images; &#x03C1;=0.32, 95% CI &#x2013;0.39 to 0.79; <italic>P</italic>=.37 with images), suggesting little alignment with human rankings.</p></sec><sec id="s3-6"><title>Evaluation of Model Performance on Unseen Questions After Knowledge Cutoff</title><p>All examination questions used in prior evaluations were published before the knowledge cutoff dates of the respective models, raising the possibility that the models may have directly encountered these questions or answer choices during training. To address this concern, we evaluated model performance using the 71st National Examination for Clinical Laboratory Technicians (administered in 2025), which consists of 200 newly released questions (100 in the morning and 100 in the afternoon) published after the respective models&#x2019; knowledge cutoffs.</p><p>As a result, GPT-5 achieved 91.5%, Gemini 2.5 Pro 90.7%, GPT-4o 82.0%, Gemini 1.5 Pro 70.5%, GPT-4o-mini 57.2%, and Neva-22B 37.5% (<xref ref-type="fig" rid="figure9">Figure 9</xref>). None of the models exhibited substantial deviations from their respective average accuracies over the previous 4 years (2020&#x2010;2023). The largest increase was observed in Neva-22B, which improved by 7.9 points compared with its historical average of 29.6%. GPT-4o-mini also showed a modest increase of 1.2 points relative to its past average of 56.0%. By contrast, GPT-5 slightly declined from 93.9% to 91.5% (&#x2013;2.4 points), and Gemini 2.5 Pro from 92.5% to 90.7% (&#x2013;1.8 points).</p><fig position="float" id="figure9"><label>Figure 9.</label><caption><p>Comparison of model accuracy before and after the knowledge cutoff. The figure shows, for each model&#x2014;GPT-5, GPT-4o, GPT-4o-mini, Neva-22B, Gemini 1.5 Pro, and Gemini 2.5 Pro&#x2014;the average accuracy from 2020&#x2010;2023 (blue) and the accuracy on the 71st National Examination for Clinical Laboratory Technicians released in 2025 (red), which was published after the knowledge cutoff. Across all models, the 2025 accuracy was comparable to or higher than the historical average, with no significant performance degradation observed. The blue dashed line indicates the average accuracy of each model across the 66th to 69th examinations. Blue points are the 4 yearly accuracies; error bars are 95% Wilson score CIs for the 2020&#x2010;2023 mean (n=800) and for 2025 (n=200).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mededu_v12i1e84266_fig09.png"/></fig><p>To statistically validate these observations, we performed chi-square tests comparing the accuracy between the training years (2020&#x2010;2023) and the 2025 examination for each model and image condition. As summarized in <xref ref-type="table" rid="table4">Table 4</xref>, none of the comparisons showed statistically significant declines (<italic>P</italic>&#x2265;.05). These results confirm that the observed declines in accuracy are not statistically significant and support the conclusion that no material performance degradation was observed in the 2025 examination. These findings indicate that all models maintained comparable accuracy when applied to previously unseen questions created after their respective knowledge cutoffs, with no significant performance degradation observed.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Chi-square test results comparing model accuracy between the training years (2020&#x2010;2023; n=800 questions) and the 2025 examination (n=200 questions). For each model and image condition, accuracy is the mean across 3 independent runs (see Methods). A 2&#x00D7;2 contingency table of correct versus incorrect responses was constructed using question-level counts. &#x0394;Acc denotes the difference in accuracy (2025 minus 2020&#x2010;2023, in percentage points), with its two-sided 95% CI computed by the Newcombe hybrid score method.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" colspan="2">Model and condition</td><td align="left" valign="bottom">2020&#x2010;2023 Acc (%)</td><td align="left" valign="bottom">2025 Acc (%)</td><td align="left" valign="bottom">&#x0394;Acc (pp)<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="bottom">95% CI of &#x0394;Acc (pp)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">GPT-4o-mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Image given</td><td align="left" valign="top">55.96</td><td align="left" valign="top">57.17</td><td align="left" valign="top">1.21</td><td align="left" valign="top">&#x2013;6.51 to 8.71</td><td align="left" valign="top">.80</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No image given</td><td align="left" valign="top">53.04</td><td align="left" valign="top">55.67</td><td align="left" valign="top">2.62</td><td align="left" valign="top">&#x2013;5.11 to 10.18</td><td align="left" valign="top">.53</td></tr><tr><td align="left" valign="top" colspan="2">GPT-4o</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Image given</td><td align="left" valign="top">81.67</td><td align="left" valign="top">82.00</td><td align="left" valign="top">0.33</td><td align="left" valign="top">&#x2013;6.10 to 5.82</td><td align="left" valign="top">.90</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No image given</td><td align="left" valign="top">79.75</td><td align="left" valign="top">79.50</td><td align="left" valign="top">&#x2013;0.25</td><td align="left" valign="top">&#x2013;6.92 to 5.55</td><td align="left" valign="top">.94</td></tr><tr><td align="left" valign="top" colspan="2">Gemini 1.5 Pro</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Image given</td><td align="left" valign="top">69.42</td><td align="left" valign="top">70.50</td><td align="left" valign="top">1.08</td><td align="left" valign="top">&#x2013;6.26 to 7.82</td><td align="left" valign="top">.76</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No image given</td><td align="left" valign="top">68.46</td><td align="left" valign="top">67.83</td><td align="left" valign="top">&#x2013;0.62</td><td align="left" valign="top">&#x2013;8.07 to 6.30</td><td align="left" valign="top">.89</td></tr><tr><td align="left" valign="top" colspan="2">Gemini 2.5 Pro</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Image given</td><td align="left" valign="top">92.50</td><td align="left" valign="top">90.67</td><td align="left" valign="top">&#x2013;1.83</td><td align="left" valign="top">&#x2013;6.93 to 2.04</td><td align="left" valign="top">.35</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No image given</td><td align="left" valign="top">86.75</td><td align="left" valign="top">87.67</td><td align="left" valign="top">0.92</td><td align="left" valign="top">&#x2013;4.79 to 5.53</td><td align="left" valign="top">.78</td></tr><tr><td align="left" valign="top" colspan="2">Neva-22B</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Image given</td><td align="left" valign="top">29.58</td><td align="left" valign="top">37.50</td><td align="left" valign="top">7.92</td><td align="left" valign="top">0.72 to 15.45</td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No image given</td><td align="left" valign="top">32.12</td><td align="left" valign="top">38.50</td><td align="left" valign="top">6.38</td><td align="left" valign="top">&#x2013;0.89 to 13.96</td><td align="left" valign="top">.09</td></tr><tr><td align="left" valign="top" colspan="2">GPT-5</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Image given</td><td align="left" valign="top">93.92</td><td align="left" valign="top">91.50</td><td align="left" valign="top">&#x2013;2.42</td><td align="left" valign="top">&#x2013;7.33 to 1.23</td><td align="left" valign="top">.23</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No image given</td><td align="left" valign="top">88.17</td><td align="left" valign="top">86.50</td><td align="left" valign="top">&#x2013;1.67</td><td align="left" valign="top">&#x2013;7.47 to 3.06</td><td align="left" valign="top">.53</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>pp: percentage point.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study systematically evaluated multiple LVLMs using Kensagishi QA, an examination-specific benchmark constructed from the Japanese National Examination for Clinical Laboratory Technicians (2020&#x2010;2023), and demonstrated that several recent cloud-hosted models exceeded the human passing threshold. At the same time, substantial differences were observed across models in overall accuracy, visual integration, and alignment with human performance patterns, clarifying their respective performance characteristics and applicability in medical contexts.</p></sec><sec id="s4-2"><title>Interpretation and Comparison With Prior Work</title><p>At a high level, recent cloud-hosted LVLMs exceeded the human passing threshold, with GPT-5 showing the strongest overall performance, Gemini 2.5 Pro close behind, and GPT-4o also performing robustly. Gemini 1.5 Pro exhibited greater variability by subfield and condition, whereas GPT-4o-mini and Neva-22B performed relatively poorly overall. Crucially, for image-based questions, providing images improved accuracy for the top-performing models (eg, GPT-5, Gemini 2.5 Pro, and GPT-4o), indicating that visual input contributes materially to performance; by contrast, Neva-22B did not show such gains. These findings suggest that the ability to integrate visual information is a key driver of accuracy in this examination setting. Importantly, this effect was not universal but model-dependent, underscoring that multimodal capability cannot be assumed even when visual input is available.</p><p>These findings are broadly consistent with previous studies showing that recent LLMs can achieve high performance on medical licensing examinations. Med-PaLM 2 demonstrated expert-level performance on MedQA and other medical question-answering benchmarks, while subsequent studies incorporating structured clinical knowledge further improved performance on USMLE-style examinations [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. In the Japanese context, GPT-4 successfully passed multiple years of the Japanese National Medical Licensing Examination using the text-based IgakuQA benchmark [<xref ref-type="bibr" rid="ref11">11</xref>]. Previous work has also reported that GPT-4-class models achieve passing-level performance on the Japanese National Examination for Clinical Laboratory Technicians [<xref ref-type="bibr" rid="ref13">13</xref>]. Our results are consistent with these reports but extend them in 3 important respects: we evaluated multiple model generations and providers under a unified protocol, explicitly quantified the contribution of visual input using paired image-provided and no-image conditions, and additionally evaluated questions released after the reported knowledge cutoffs.</p><p>The paired image-provided and no-image evaluation also provides insights that cannot be obtained from conventional text-only examination benchmarks. Previous multimodal medical AI studies have demonstrated that vision-language models can answer biomedical image questions [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]; however, relatively little has been reported regarding the extent to which visual input contributes to performance on professional licensing examinations. In our benchmark, image provision substantially improved performance for GPT-5, Gemini 2.5 Pro, and GPT-4o, whereas Neva-22B showed no benefit. These findings indicate that multimodal capability should not be regarded as a binary property. Instead, the effectiveness of visual information integration differs markedly among models, even when all models technically accept image input.</p><p>Subfield-level patterns further suggest that performance reflects not only overall model capacity but also the cognitive demands of each subfield. For example, the relatively strong performance of smaller models in general clinical laboratory science may reflect its emphasis on standardized procedures and conceptual knowledge, whereas visually intensive or quantitatively demanding subfields (eg, pathological histocytology and clinical chemistry) showed larger gaps between high- and lower-capacity models.</p><p>Beyond absolute accuracy, we examined whether models mirrored human domain-level patterns. GPT-4o (and, to a lesser extent, GPT-4o-mini) displayed relatively strong rank alignment with human subfield trends, while GPT-5 achieved the highest accuracy with more limited alignment. Gemini 1.5 Pro showed the strongest human alignment when images were withheld but weaker alignment with images provided. This divergence suggests potentially different use cases; however, strong alignment with human difficulty patterns does not necessarily imply superior pedagogical utility, as it may also reflect shared error tendencies rather than improved explanatory capacity.</p><p>To probe generalizability beyond potential training-set exposure, we evaluated all models on the 71st examination (administered in 2025), released after each model&#x2019;s knowledge cutoff. Accuracies were comparable to historical averages from 2020&#x2010;2023, indicating no substantial degradation on unseen questions. This suggests that the evaluated LVLMs leverage transferable knowledge and reasoning rather than memorization of specific items and that the examination&#x2019;s structure and content remain sufficiently stable for such capabilities to carry over.</p><p>For practical deployment, the implications of the present findings are primarily educational rather than clinical. Kensagishi QA provides a standardized benchmark for comparing cloud-hosted LVLMs before their adoption in examination preparation systems or educational support tools. The marked effect of visual input also suggests that evaluations intended for visually intensive subjects should include the original examination images rather than relying on text-only transcriptions. At the same time, our study evaluated only answer selection and did not assess explanation quality, hallucinations, calibration, or educational effectiveness [<xref ref-type="bibr" rid="ref28">28</xref>]. Therefore, high examination accuracy alone should not be interpreted as sufficient evidence for deployment in either educational or clinical settings without additional validation.</p></sec><sec id="s4-3"><title>Limitations</title><p>Several limitations should be considered. First, our evaluation focused exclusively on multiple-choice accuracy and did not assess reliability-related dimensions such as factuality, instruction following, safety [<xref ref-type="bibr" rid="ref28">28</xref>-<xref ref-type="bibr" rid="ref30">30</xref>], or appropriate handling of Japanese medical terminology [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Second, all experiments were conducted under a single &#x201C;numbers-only&#x201D; direct-answer prompting condition, which may not fully elicit multi-step reasoning. Prior work has reported that reasoning-oriented prompting strategies, including chain-of-thought, can improve performance in some medical and complex reasoning tasks [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. Therefore, the reported accuracies should be interpreted as standardized baseline performance under controlled prompting conditions rather than prompt-optimized upper-bound estimates. Third, the number of image-based questions (n=130) was relatively small, limiting statistical precision in estimating multimodal effects and model-by-image interactions. Fourth, our evaluation did not include high-capacity or medically specialized open-weight LVLMs [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]; thus, the present results should not be interpreted as a direct comparison between open- and closed-weight models.</p></sec><sec id="s4-4"><title>Conclusions</title><p>Looking ahead, we plan to broaden coverage to additional model families (including high-capacity and medically specialized open-weight, self-hosted medical LVLMs) under a common protocol; incorporate multidimensional quality metrics that combine accuracy with factuality, instruction compliance, safety, and terminology handling [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]; refine prompt design to better leverage domain knowledge [<xref ref-type="bibr" rid="ref33">33</xref>]; and investigate the interplay between retrieval-augmented approaches and intrinsic reasoning for tasks such as differential diagnosis. Systematic comparison between cloud-hosted and open-weight LVLMs will be necessary to determine whether observed performance differences reflect deployment modality, model scale, architectural design, or other factors. Ultimately, these efforts will inform the development of learning-support tools that visualize mastery, target weak areas, and provide explanatory feedback for national examination preparation and broader medical education [<xref ref-type="bibr" rid="ref34">34</xref>]. In summary, by providing the first systematic, multi-model evaluation of cloud-hosted LVLMs on this examination with explicit isolation of visual input&#x2014;extending prior text-only, few-model studies&#x2014;this work delivers a reproducible Kensagishi QA benchmark and actionable, use-case&#x2013;specific guidance for deploying LVLMs in medical education and clinical training. More broadly, this framework illustrates how nationally specific professional examinations can serve as rigorous, multilingual testbeds for evaluating and improving medical AI systems in realistic, high-stakes contexts.</p></sec></sec></body><back><ack><p>The authors declare the use of generative AI in the research and writing process. According to the GAIDeT taxonomy (2025), the following tasks were delegated to generative AI (GenAI) tools under full human supervision: proofreading and editing, and translation. The GenAI tool used was ChatGPT (GPT-4o, o3, and GPT-4.5) (OpenAI). ChatGPT was used solely for English-language editing&#x2014;translation, grammar correction, and stylistic refinement&#x2014;applied to individual sentences and paragraphs already written by the authors. It was not used to generate new content, summarize, analyze or interpret data, or modify any scientific conclusions. All authors reviewed and edited the AI-assisted text and take full responsibility for the accuracy, originality, and integrity of the manuscript. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes. This declaration is submitted on behalf of all authors (collective responsibility).</p></ack><notes><sec><title>Funding</title><p>This work was supported by the JST NBDC Integrated Database Promotion Program (JPMJND2402 to HO) and the RIKEN TRIP Advanced General Intelligence for Science Program (AGIS) (to HO).</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: KM, HO</p><p>Data curation: KM</p><p>Formal analysis: KM</p><p>Funding acquisition: HO</p><p>Investigation: KM</p><p>Methodology: KM, RM, YT-A, HO</p><p>Project administration: HO</p><p>Resources: KM</p><p>Software: KM</p><p>Supervision: RM, YT-A, HO</p><p>Validation: KM</p><p>Visualization: KM</p><p>Writing &#x2013; original draft: KM, RM, YT-A, HO</p><p>Writing &#x2013; review &#x0026; editing: KM, RM, YT-A, HO</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">GPU</term><def><p>graphics processing unit</p></def></def-item><def-item><term id="abb2">Kensagishi QA</term><def><p>Kensagishi question-answer</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb4">LVLM</term><def><p>large vision-language model</p></def></def-item><def-item><term id="abb5">MHLW</term><def><p>Ministry of Health, Labour and Welfare</p></def></def-item><def-item><term id="abb6">PDL1.0</term><def><p>Public Data License version 1.0</p></def></def-item><def-item><term id="abb7">SCAI</term><def><p>Semantic Clinical AI</p></def></def-item><def-item><term id="abb8">USMLE</term><def><p>United States Medical Licensing Examination</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Mann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ryder</surname><given-names>N</given-names> </name><name name-style="western"><surname>Subbiah</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kaplan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dhariwal</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Language models are few-shot learners</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 22, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>OpenAI</collab><name name-style="western"><surname>Adler</surname><given-names>S</given-names> </name><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ahmad</surname><given-names>L</given-names> </name><name name-style="western"><surname>Akkaya</surname><given-names>I</given-names> </name><etal/></person-group><article-title>GPT-4 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 15, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.08774</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shao</surname><given-names>M</given-names> </name><name name-style="western"><surname>Basit</surname><given-names>A</given-names> </name><name name-style="western"><surname>Karri</surname><given-names>R</given-names> </name><name name-style="western"><surname>Shafique</surname><given-names>M</given-names> </name></person-group><article-title>Survey of different large language model architectures: trends, benchmarks, and challenges</article-title><source>IEEE Access</source><year>2024</year><volume>12</volume><fpage>188664</fpage><lpage>188706</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2024.3482107</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>J</given-names> </name><etal/></person-group><article-title>MM-LLMs: recent advances in multimodal large language models</article-title><access-date>2026-08-13</access-date><conf-name>Findings of the Association for Computational Linguistics ACL 2024</conf-name><conf-date>Aug 11-16, 2024</conf-date><conf-loc>Bangkok, Thailand and virtual meeting. 2024</conf-loc><fpage>12401</fpage><lpage>12430</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.findings-acl.738/">https://aclanthology.org/2024.findings-acl.738/</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.738</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dunlap</surname><given-names>L</given-names> </name><name name-style="western"><surname>Mashita</surname><given-names>K</given-names> </name><etal/></person-group><article-title>VisionArena: 230K real world user-VLM conversations with preference labels</article-title><conf-name>2025 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name><conf-date>Jun 11-15, 2025</conf-date><conf-loc>Nashville, TN, USA. 2025</conf-loc><fpage>3877</fpage><lpage>3887</lpage><pub-id pub-id-type="doi">10.1109/CVPR52734.2025.00367</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Are we on the right way for evaluating large vision-language models?</article-title><access-date>2026-08-13</access-date><conf-name>Advances in Neural Information Processing Systems 37</conf-name><conf-date>Dec 10-15, 2024</conf-date><conf-loc>Vancouver, BC, Canada. 2024</conf-loc><fpage>27056</fpage><lpage>27087</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://www.proceedings.com/79017.html">http://www.proceedings.com/79017.html</ext-link></comment><pub-id pub-id-type="doi">10.52202/079017-0850</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Oufattole</surname><given-names>N</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Szolovits</surname><given-names>P</given-names> </name></person-group><article-title>What disease does this patient have? A large-scale open domain question answering dataset from medical exams</article-title><source>Appl Sci</source><year>2021</year><volume>11</volume><issue>14</issue><fpage>6421</fpage><pub-id pub-id-type="doi">10.3390/app11146421</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>H</given-names> </name><etal/></person-group><article-title>LLM-MedQA: enhancing medical question answering through case studies in large language models</article-title><conf-name>2025 International Joint Conference on Neural Networks (IJCNN)</conf-name><conf-date>Jun 30 to Jul 5, 2025</conf-date><pub-id pub-id-type="doi">10.1109/IJCNN64981.2025.11228647</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elkin</surname><given-names>PL</given-names> </name><name name-style="western"><surname>Mehta</surname><given-names>G</given-names> </name><name name-style="western"><surname>LeHouillier</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Semantic clinical artificial intelligence vs native large language model performance on the USMLE</article-title><source>JAMA Netw Open</source><year>2025</year><month>04</month><day>1</day><volume>8</volume><issue>4</issue><fpage>e256359</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.6359</pub-id><pub-id pub-id-type="medline">40261653</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kasai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kasai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sakaguchi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yamada</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Radev</surname><given-names>D</given-names> </name></person-group><article-title>Evaluating GPT-4 and ChatGPT on Japanese medical licensing examinations</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 31, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.18027</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kido</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yamada</surname><given-names>H</given-names> </name><name name-style="western"><surname>Tokunaga</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Evaluation of LLM-generated distractors of multiple-choice questions for the Japanese National Nursing Examination</article-title><access-date>2026-08-13</access-date><conf-name>17th International Conference on Computer Supported Education</conf-name><conf-date>Apr 1-3, 2025</conf-date><conf-loc>Porto, Portugal. 2025</conf-loc><fpage>754</fpage><lpage>764</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://www.scitepress.org/DigitalLibrary/ProceedingLink.aspx?ID=1900">http://www.scitepress.org/DigitalLibrary/ProceedingLink.aspx?ID=1900</ext-link></comment><pub-id pub-id-type="doi">10.5220/0013460300003932</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Doi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ishida</surname><given-names>H</given-names> </name><name name-style="western"><surname>Nagasawa</surname><given-names>H</given-names> </name><name name-style="western"><surname>Tsuboi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kikuchi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ichino</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Performance of generative pretrained transformer on the national licensing examination for medical technologist in Japan</article-title><source>Jpn J Med Technol</source><year>2024</year><volume>73</volume><issue>2</issue><fpage>323</fpage><lpage>331</lpage><pub-id pub-id-type="doi">10.14932/jamt.23-80</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ohshige</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ninomiya</surname><given-names>A</given-names> </name><name name-style="western"><surname>Akaza</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Can large language models pass the national examination for medical technologists?</article-title><source>Jpn J Med Technol Educ</source><year>2024</year><access-date>2026-08-31</access-date><volume>16</volume><fpage>99</fpage><lpage>105</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.nitirinkyo.jp/cms2025/wp-content/uploads/2024/09/magazine1602_02.pdf">https://www.nitirinkyo.jp/cms2025/wp-content/uploads/2024/09/magazine1602_02.pdf</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tanaka</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Nakata</surname><given-names>T</given-names> </name><name name-style="western"><surname>Aiga</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Performance of generative pretrained transformer on the National Medical Licensing Examination in Japan</article-title><source>PLoS Digit Health</source><year>2024</year><month>01</month><volume>3</volume><issue>1</issue><fpage>e0000433</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000433</pub-id><pub-id pub-id-type="medline">38261580</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><article-title>Questions and correct answers for the 66th national examination for medical technologists</article-title><source>Ministry of Health, Labour and Welfare</source><year>2020</year><access-date>2025-06-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp200414-07.html">https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp200414-07.html</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>Questions and correct answers for the 67th national examination for medical technologists</article-title><source>Ministry of Health, Labour and Welfare</source><year>2021</year><access-date>2025-06-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp210416-07.html">https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp210416-07.html</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="web"><article-title>Questions and correct answers for the 68th national examination for medical technologists</article-title><source>Ministry of Health, Labour and Welfare</source><year>2022</year><access-date>2025-06-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp220421-07.html">https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp220421-07.html</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="web"><article-title>Questions and correct answers for the 69th national examination for medical technologists</article-title><source>Ministry of Health, Labour and Welfare</source><year>2023</year><access-date>2025-06-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp230524-07.html">https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp230524-07.html</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Onodera</surname><given-names>R</given-names> </name></person-group><article-title>An approach based on national examination results analysis: Clinical laboratory technologist education and postgraduate education and qualifications required after the curriculum revision</article-title><source>Jpn J Med Technol Educ</source><year>2024</year><access-date>2026-09-02</access-date><volume>16</volume><fpage>37</fpage><lpage>44</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.nitirinkyo.jp/cms2025/wp-content/uploads/2024/03/magazine1601_06.pdf">https://www.nitirinkyo.jp/cms2025/wp-content/uploads/2024/03/magazine1601_06.pdf</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wilson</surname><given-names>EB</given-names> </name></person-group><article-title>Probable inference, the law of succession, and statistical inference</article-title><source>J Am Stat Assoc</source><year>1927</year><month>06</month><volume>22</volume><issue>158</issue><fpage>209</fpage><lpage>212</lpage><pub-id pub-id-type="doi">10.1080/01621459.1927.10502953</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Newcombe</surname><given-names>RG</given-names> </name></person-group><article-title>Interval estimation for the difference between independent proportions: Comparison of eleven methods</article-title><source>Stat Med</source><year>1998</year><volume>17</volume><issue>8</issue><fpage>873</fpage><lpage>890</lpage><pub-id pub-id-type="doi">10.1002/(SICI)1097-0258(19980430)17:8&#x003C;873::AID-SIM779&#x003E;3.0.CO;2-I</pub-id><pub-id pub-id-type="medline">9595617</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Murakami</surname><given-names>K</given-names> </name><name name-style="western"><surname>Matsuzawa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tahara-Arai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ozaki</surname><given-names>H</given-names> </name></person-group><article-title>KensagishiQA: Benchmark data for evaluating large language models on the Japanese National Examination for Clinical Laboratory Technicians</article-title><source>GitHub</source><year>2026</year><month>07</month><day>15</day><access-date>2026-09-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/bioinfo-tsukuba/KensagishiQA">https://github.com/bioinfo-tsukuba/KensagishiQA</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><article-title>Questions and correct answers for the 71st national examination for medical technologists</article-title><source>Ministry of Health, Labour and Welfare</source><year>2025</year><access-date>2025-07-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp250428-07.html">https://www.mhlw.go.jp/seisakunitsuite/bunya/kenkou_iryou/iryou/topics/tp250428-07.html</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="web"><article-title>Ethical guidelines for medical and biological research involving human subjects</article-title><source>Ministry of Education, Culture, Sports, Science and Technology, Ministry of Health, Labour and Welfare, Ministry of Economy, Trade and Industry</source><year>2023</year><access-date>2026-07-18</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/content/001457376.pdf">https://www.mhlw.go.jp/content/001457376.pdf</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><etal/></person-group><article-title>LLaVA-Med: training a large language-and-vision assistant for biomedicine in one day</article-title><access-date>2026-08-13</access-date><conf-name>Advances in Neural Information Processing Systems 36</conf-name><conf-date>Dec 10-16, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="http://www.proceedings.com/75280.html">http://www.proceedings.com/75280.html</ext-link></comment><pub-id pub-id-type="doi">10.52202/075280-1240</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Moor</surname><given-names>M</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yasunaga</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zakka</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dalmia</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Med-Flamingo: a multimodal medical few-shot learner</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 27, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2307.15189</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Asgari</surname><given-names>E</given-names> </name><name name-style="western"><surname>Monta&#x00F1;a-Brown</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dubois</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>13</day><volume>8</volume><issue>1</issue><fpage>274</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01670-7</pub-id><pub-id pub-id-type="medline">40360677</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="medline">37438534</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Niu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xun</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hou</surname><given-names>W</given-names> </name><etal/></person-group><article-title>JMedEthicBench: a multi-turn conversational benchmark for evaluating medical safety in Japanese large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 4, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2601.01627</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Aizawa</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Rambow</surname><given-names>O</given-names> </name><name name-style="western"><surname>Wanner</surname><given-names>L</given-names> </name><name name-style="western"><surname>Apidianaki</surname><given-names>M</given-names> </name><name name-style="western"><surname>Al-Khalifa</surname><given-names>H</given-names> </name><name name-style="western"><surname>Eugenio</surname><given-names>BD</given-names> </name><name name-style="western"><surname>Schockaert</surname><given-names>S</given-names> </name></person-group><article-title>JMedBench: a benchmark for evaluating Japanese biomedical large language models</article-title><access-date>2026-08-31</access-date><conf-name>Proceedings of the 31st international conference on computational linguistics</conf-name><conf-date>Jan 19-24, 2025</conf-date><conf-loc>Abu Dhabi, UAE. 2025</conf-loc><fpage>5918</fpage><lpage>5935</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.coling-main.395/">https://aclanthology.org/2025.coling-main.395/</ext-link></comment></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Schuurmans</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Chain-of-thought prompting elicits reasoning in large language models</article-title><access-date>2026-08-13</access-date><conf-name>Advances in Neural Information Processing Systems 35</conf-name><conf-date>Dec 6-9, 2022</conf-date><conf-loc>New Orleans, Louisiana. 2022</conf-loc><fpage>24824</fpage><lpage>24837</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://www.proceedings.com/68431.html">http://www.proceedings.com/68431.html</ext-link></comment><pub-id pub-id-type="doi">10.52202/068431-1800</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Prompt engineering in consistency and reliability with the evidence-based guideline for LLMs</article-title><source>NPJ Digit Med</source><year>2024</year><month>02</month><day>20</day><volume>7</volume><issue>1</issue><fpage>41</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01029-4</pub-id><pub-id pub-id-type="medline">38378899</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ch&#x2019;en</surname><given-names>PY</given-names> </name><name name-style="western"><surname>Day</surname><given-names>W</given-names> </name><name name-style="western"><surname>Pekson</surname><given-names>RC</given-names> </name><etal/></person-group><article-title>GPT-4 generated answer rationales to multiple choice assessment questions in undergraduate medical education</article-title><source>BMC Med Educ</source><year>2025</year><month>03</month><day>4</day><volume>25</volume><issue>1</issue><fpage>333</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-06862-z</pub-id><pub-id pub-id-type="medline">40038669</pub-id></nlm-citation></ref></ref-list></back></article>