<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e95859</article-id><article-id pub-id-type="doi">10.2196/95859</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Performance and Hallucination Analysis of Large Language Models on European Anesthesiology Examinations: Cross-Sectional Comparative Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Andrei</surname><given-names>Stefan</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Giet</surname><given-names>Thibault</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Belouard</surname><given-names>Alexis</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Stefan</surname><given-names>Mihai</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Popescu</surname><given-names>Mihai</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tanaka</surname><given-names>S&#x00E9;bastien</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Montravers</surname><given-names>Philippe</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gouel</surname><given-names>Aur&#x00E9;lie</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff8">8</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Anesthesiology and Intensive Care, CHU Bichat Claude Bernard, Assistance Publique, H&#x00F4;pitaux de Paris</institution><addr-line>Paris</addr-line><addr-line>&#x00CE;le-de-France</addr-line><country>France</country></aff><aff id="aff2"><institution>Group of Data Modeling, Computational Biology, and Predictive Medicine, Applied Mathematics, Institut de Biologie de l'&#x00C9;cole Normale Sup&#x00E9;rieure</institution><addr-line>Paris</addr-line><addr-line>&#x00CE;le-de-France</addr-line><country>France</country></aff><aff id="aff3"><institution>Discipline of Anaesthesiology and Intensive Care, Carol Davila University of Medicine and Pharmacy</institution><addr-line>Sos Fundeni, 258</addr-line><addr-line>Bucharest</addr-line><addr-line>Bucure&#x0219;ti</addr-line><country>Romania</country></aff><aff id="aff4"><institution>2nd of Department of Anaesthesiology and Intensive Care, Institutul de Urgen&#x0163;&#x0103; pentru Boli Cardiovasculare "Prof.Dr. C.C. Iliescu"</institution><addr-line>Bucharest</addr-line><addr-line>Bucure&#x0219;ti</addr-line><country>Romania</country></aff><aff id="aff5"><institution>Department of Anaesthesiology and Intensive Care, Bucharest Emergency University Hospital</institution><addr-line>Bucharest</addr-line><addr-line>Bucure&#x0219;ti</addr-line><country>Romania</country></aff><aff id="aff6"><institution>Diab&#x00E8;te ath&#x00E9;rothrombose Th&#x00E9;rapies R&#x00E9;union Oc&#x00E9;an Indien (D&#x00E9;TROI), INSERM UMR 1188, University of Reunion Island</institution><addr-line>Saint-Pierre de La R&#x00E9;union</addr-line><addr-line>R&#x00E9;union</addr-line><country>R&#x00E9;union</country></aff><aff id="aff7"><institution>Physiopathology and Epidemiology of Respiratory Diseases, INSERM UMR1152, Universit&#x00E9; Paris Cit&#x00E9;</institution><addr-line>Paris</addr-line><addr-line>&#x00CE;le-de-France</addr-line><country>France</country></aff><aff id="aff8"><institution>Antibodies in Therapy and Pathology, Institut Pasteur, INSERM UMR1222, Universit&#x00E9; Paris Cit&#x00E9;</institution><addr-line>Paris</addr-line><addr-line>&#x00CE;le-de-France</addr-line><country>France</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>MacNeill</surname><given-names>Luke</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Kasagga</surname><given-names>Alousious</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Cetin</surname><given-names>Niyazi</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Mihai Stefan, MD, Discipline of Anaesthesiology and Intensive Care, Carol Davila University of Medicine and Pharmacy, Sos Fundeni, 258, Bucharest, Bucure&#x0219;ti, 022322, Romania, 40 742185533; <email>mihai.stefan@umfcd.ro</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>1</day><month>9</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e95859</elocation-id><history><date date-type="received"><day>22</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>23</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Stefan Andrei, Thibault Giet, Alexis Belouard, Mihai Stefan, Mihai Popescu, S&#x00E9;bastien Tanaka, Philippe Montravers, Aur&#x00E9;lie Gouel. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 1.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e95859"/><abstract><sec><title>Background</title><p>Large language models (LLMs) have shown promising performance on medical examinations across specialties. However, comparative evaluations of current-generation LLMs across multiple European anesthesiology examinations, alongside structured assessment of hallucinations vs question-related confusion, remain lacking.</p></sec><sec><title>Objective</title><p>This study aimed to compare the performance of 4 state-of-the-art LLMs on anesthesiology and intensive medicine examination questions and assess their hallucination rates.</p></sec><sec sec-type="methods"><title>Methods</title><p>This computational comparative study analyzed 437 multiple-choice questions (1748 queries) from 3 sources: nurse anesthetist school examinations (infirmier anesth&#x00E9;siste dipl&#x00F4;m&#x00E9; d&#x2019;&#x00C9;tat [registered nurse anesthetist]; n=100, 22.9%), European Diploma in Anaesthesiology and Intensive Care (EDAIC; n=219, 50.1%), and EDAIC On-Line Assessment (n=118, 27.0%). Each question was submitted to 4 LLMs (Claude Sonnet 4.5, Gemini 2.5 Pro, GPT-5, and Grok 4) using standardized prompts via default web interface settings. Responses were evaluated through structured consensus review by 2 examiners for accuracy, hallucinations, and question-related confusion. Statistical analysis included Friedman and Wilcoxon signed-rank tests with Holm-Bonferroni correction, the Cochran <italic>Q</italic> test, and generalized estimating equations.</p></sec><sec sec-type="results"><title>Results</title><p>Average success rates ranged from 86% (SD 18%) to 94% (SD 10%) across LLMs and examination types, exceeding the EDAIC part I passing threshold, representing substantial improvement over previously reported GPT-3.5 performance. For the EDAIC, overall intermodel differences were significant (Friedman <italic>&#x03C7;</italic><sup>2</sup><sub>3</sub>=13.9; <italic>P</italic>=.003; <italic>W</italic>=0.02), with Gemini outperforming GPT-5 as the only pairwise difference. Hallucination rates ranged from 11% (11/100) to 20.1% (44/219) without significant intermodel differences. All models exceeded the EDAIC passing threshold.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Current-generation LLMs demonstrated consistently high performance across multiple European anesthesiology examinations but continue to produce clinically relevant hallucinations, supporting their role as supervised educational tools rather than autonomous learning resources. These findings underscore the need for structured integration frameworks and systematic verification when deploying LLMs as learning tools in medical education.</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>large language models</kwd><kwd>medical education</kwd><kwd>anesthesiology</kwd><kwd>intensive care medicine</kwd><kwd>hallucinations</kwd><kwd>examination</kwd><kwd>multiple-choice questions</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Large language models (LLMs) have gained increasing attention in health care and medical education due to their demonstrated capacity to generate clinically relevant text, answer medical questions, and support reasoning tasks. Benchmark analyses using different standardized assessments have shown that LLMs can achieve substantial accuracy, sometimes approaching or surpassing human performance [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. However, empirical evidence also reveals wide variability across tasks and domains, with persistent limitations in multimodal reasoning, complex clinical scenarios, and domain-dependent weaknesses, raising concerns about the consistency, precision, and clinical applicability of these systems [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>LLMs may support the preparation of future health care professionals for clinical environments where AI-assisted reasoning will become increasingly prevalent. These competencies may become more relevant for modern medical practice [<xref ref-type="bibr" rid="ref6">6</xref>]. While no universally accepted definition exists, the French National Commission on Informatics and Liberty describes LLMs as statistical models of linguistic unit distribution that predict subsequent words in a sequence [<xref ref-type="bibr" rid="ref7">7</xref>]. This text completion mechanism, trained on vast datasets including Wikipedia, internet sources, and books, enables coherent text generation but also introduces the potential for error propagation from training data [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Despite enthusiasm regarding their potential, emerging evidence indicates that LLMs exhibit systematic vulnerabilities, including hallucinations, overconfidence, and errors in complex reasoning tasks. Operating on probabilistic text completion rather than genuine reasoning or comprehension [<xref ref-type="bibr" rid="ref9">9</xref>], LLMs generate plausible-appearing text with perfect syntax and grammar that may diverge from contextual accuracy, user instructions, or factual knowledge [<xref ref-type="bibr" rid="ref7">7</xref>]. These errors, collectively termed &#x201C;hallucinations,&#x201D; are defined by Huang et al [<xref ref-type="bibr" rid="ref10">10</xref>] as phenomena where generated content appears nonsensical or unfaithful to the source material. Furthermore, analyses of LLM performance across cognitive domains suggest that, while these models perform adequately in factual recall and substantial clinical knowledge, their accuracy declines in tasks requiring higher-order reasoning and clinical judgment [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>In a discipline requiring high-level expertise and practical knowledge, evaluating whether LLMs meet sufficient standards for pedagogical use is imperative. This study aimed to assess LLM performance and their assessment potential in anesthesiology education using standard examination questions. To our knowledge, no study has simultaneously compared multiple current-generation LLMs across different European anesthesiology examinations while systematically distinguishing hallucinations from question-related confusion.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This computational comparative study evaluated 4 LLMs: Claude Sonnet 4.5 (Anthropic), Gemini 2.5 Pro (Google), GPT-5 (OpenAI), and Grok 4 (xAI), accessed via their respective publicly available web interfaces in October 2025. Models were accessed using default parameters (temperature and other generation settings were not modified). Exact internal model checkpoints were not disclosed by model providers.</p></sec><sec id="s2-2"><title>Objectives and End Points</title><p>The primary objective was to comparatively evaluate LLM performance on anesthesiology examination questions. The primary end point was the percentage of correct responses across the 4 LLMs. Secondary objectives included assessment of hallucination rates and evaluation of question clarity through LLM-identified confusion. Secondary end points comprised the proportion of serious errors (hallucinations) and the proportion of questions flagged as &#x201C;confusing&#x201D; by the LLMs.</p></sec><sec id="s2-3"><title>Questions and Selection</title><p>Questions were selected using a multidimensional approach drawing from (1) nurse anesthetist school examinations from Assistance Publique - H&#x00F4;pitaux de Paris, designated as infirmier anesth&#x00E9;siste dipl&#x00F4;m&#x00E9; d&#x2019;&#x00E9;tat (IADE; in English state-registered nurse anesthetist) questions; (2) European Diploma in Anaesthesiology and Intensive Care (EDAIC) examinations; and (3) EDAIC On-Line Assessment (OLA) practice examinations. All questions were available in online public repositories from prior examinations. None of the study authors were involved in the creation or selection of IADE examination questions.</p><p>The EDAIC, offered by the European Society of Anaesthesiology and Intensive Care, comprises written and oral components. The written examination includes 60 basic science multiple-choice questions (MCQs) followed by 60 clinical MCQs covering internal and emergency medicine; general and regional anesthesia; and specialized anesthesia, including pain management, resuscitation, and intensive care [<xref ref-type="bibr" rid="ref13">13</xref>]. The OLA follows identical formatting for examination preparation. All questions were exclusively multiple choice to facilitate objective evaluation as MCQ responses are less subject to examiner subjectivity [<xref ref-type="bibr" rid="ref14">14</xref>]. The IADE examination is the qualifying examination for French state-registered nurse anesthetists. The selected questions consisted of MCQs covering 5 domains (physiology, pharmacology, anesthetic techniques and implementation, procedure-related considerations, and patient-specific factors) and were administered in French, whereas EDAIC and OLA questions were administered in English. Questions were analyzed in their original language without translation.</p></sec><sec id="s2-4"><title>Study Protocol</title><p>Questions were submitted identically to the 4 LLMs following a standardized protocol. All models were queried using default web interface settings. Each question comprised a stem with 4 to 6 possible responses, with correct answers ranging from none to all options. A new conversation thread was initiated for each query to prevent cross-contamination of responses. Each individual answer option was scored as correct or incorrect, yielding a percentage success score for each question (ie, the proportion of correctly classified options out of all available options per question). This item-level scoring approach was chosen to capture partial knowledge and provide a more granular assessment of model performance than binary whole-question scoring. This scoring approach mirrors the official EDAIC examination marking scheme, in which each individual option is independently scored as correct or incorrect and partial credit is awarded accordingly.</p><p>The standardized prompt template was as follows: &#x201C;[Question stem verbatim] Options: A) [option A] B) [option B] C) [option C] D) [option D] [E) and F) if applicable]. Please identify all correct answers and explain your reasoning.&#x201D; This format was maintained consistently across all queries to ensure comparability.</p><p>All responses were evaluated through a structured consensus process by 2 examiners (TG and AB), who jointly reviewed each response, discussed discrepancies, and reached agreement on classification. When an LLM achieved less than 100% accuracy, joint review was conducted to distinguish between question-related confusion and true hallucination [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>A hallucination was operationally defined as a response containing factually incorrect, fabricated, or unsupported medical information despite apparent comprehension of the question stem. In contrast, question-related confusion was defined as an incorrect response associated with misinterpretation of ambiguous or complex wording despite otherwise medically coherent reasoning. Representative examples illustrating these classification criteria are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>Hallucination and confusion analysis was performed for IADE and EDAIC questions only. The OLA was included as an external validation set for the primary end point (comparative success rates). Structured dual-reviewer classification of hallucinations and confusion for the OLA set would have required evaluation of 472 additional responses (118 &#x00D7; 4); the IADE and EDAIC provided 319 questions (1276 evaluated responses) across 2 distinct cognitive levels, which was considered sufficient to characterize error patterns. Given the fact that the OLA is not conceptually different from the EDAIC (it is an EDAIC simulation), it was retained as an external validation dataset for performance assessment rather than for detailed error pattern characterization.</p></sec><sec id="s2-5"><title>Statistical Analysis</title><p>Statistical analyses were performed using SPSS (version 29.0; IBM Corp) and Python (version 3.11; Python Software Foundation) with the <italic>statsmodels</italic> package (version 0.14), with the significance threshold fixed at a <italic>P</italic> value below .05 for all analyses. Quantitative variables are presented as means and SDs with 95% CIs where appropriate; qualitative variables are presented as counts and percentages. Normal distribution was assessed visually using histograms and Shapiro-Wilk testing. Results were analyzed by examination group (IADE, EDAIC, and OLA). Paired tests included Friedman and Wilcoxon signed-rank tests for quantitative variables and the Cochran <italic>Q</italic> test for qualitative variables. Pairwise post hoc comparisons following significant Friedman tests were performed using Wilcoxon signed-rank tests with exclusion of zero differences, and effect sizes were reported as <italic>r</italic> = <italic>Z</italic>/&#x221A;<italic>n</italic>. To control the familywise error rate across the 6 pairwise comparisons per examination group while avoiding the excessive conservativeness of the Bonferroni method with correlated comparisons, the Holm-Bonferroni sequential procedure was applied. Holm-adjusted <italic>P</italic> values (denoted as &#x201C;adjusted <italic>P</italic>&#x201D;) are reported for all pairwise post hoc and domain-level comparisons and were compared against the fixed .05 significance threshold; unadjusted <italic>P</italic> values are reported only for omnibus tests, to which no multiplicity correction applies. The Kendall <italic>W</italic> was reported as the effect size for the Friedman and Cochran <italic>Q</italic> tests. To account for the clustered structure of the binary outcomes (each question was answered by the 4 models), pairwise intermodel comparisons of hallucination and confusion rates were performed using generalized estimating equations (GEE) with a logit link, a binomial distribution, an exchangeable working correlation structure, and robust SEs, with questions specified as clusters. All 6 pairwise model contrasts were estimated for each outcome, and the resulting <italic>P</italic> values were Holm adjusted.</p></sec><sec id="s2-6"><title>Ethical Considerations</title><p>This computational study did not involve human participants, patient data, biological material, or identifiable personal information. This study exclusively analyzed publicly available examination questions and AI-generated responses. According to French regulations governing biomedical research (Loi n&#x00B0; 2012-300 du 5 mars 2012, relative aux recherches impliquant la personne humaine, &#x201C;Loi Jard&#x00E9;,&#x201C; modified by Ordonnance n&#x00B0; 2016-800 du 16 juin 2016 [<xref ref-type="bibr" rid="ref15">15</xref>]), research not involving human participants, identifiable personal data, or biological samples falls outside the scope of mandatory ethics committee review. The Commission Nationale de l&#x2019;Informatique et des Libert&#x00E9;s (CNIL) similarly does not require declaration for research that does not process personal data. Ethics committee approval was therefore not required for the present study. No patient data were collected, stored, or analyzed at any stage of the study.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>A total of 437 questions (1748 queries) were analyzed: 100 (22.9%) IADE questions, 219 (50.1%) EDAIC questions from 2 separate examinations, and 118 (27.0%) OLA questions. Complete results are shown in <xref ref-type="table" rid="table1">Tables 1</xref><xref ref-type="table" rid="table2"/><xref ref-type="table" rid="table3"/>-<xref ref-type="table" rid="table4">4</xref> and <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Success rates and hallucination and confusion rates by examination type.<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Examination and variables</td><td align="left" valign="bottom">Claude Sonnet 4.5</td><td align="left" valign="bottom">Gemini 2.5 Pro</td><td align="left" valign="bottom">GPT-5</td><td align="left" valign="bottom">Grok 4</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">IADE<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> (n=100 questions)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Success rate (%), mean (SD)</td><td align="left" valign="top">93 (12)</td><td align="left" valign="top">91 (13)</td><td align="left" valign="top">91 (13)</td><td align="left" valign="top">92 (14)</td><td align="left" valign="top">.16</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hallucinations, n (%)</td><td align="left" valign="top">11 (11)</td><td align="left" valign="top">16 (16)</td><td align="left" valign="top">13 (13)</td><td align="left" valign="top">16 (16)</td><td align="left" valign="top">.34</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Confusion, n (%)</td><td align="left" valign="top">8 (8)</td><td align="left" valign="top">13 (13)</td><td align="left" valign="top">14 (14)</td><td align="left" valign="top">8 (8)</td><td align="left" valign="top">.049</td></tr><tr><td align="left" valign="top" colspan="6">EDAIC<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> (n=219 questions)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Success rate (%), mean (SD)</td><td align="left" valign="top">89 (16)</td><td align="left" valign="top">89 (17)</td><td align="left" valign="top">86 (18)</td><td align="left" valign="top">87 (17)</td><td align="left" valign="top">.003</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hallucinations, n (%)</td><td align="left" valign="top">36 (16.4)</td><td align="left" valign="top">31 (14.2)</td><td align="left" valign="top">40 (18.3)</td><td align="left" valign="top">44 (20.1)</td><td align="left" valign="top">.10</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Confusion, n (%)</td><td align="left" valign="top">51 (23.3)</td><td align="left" valign="top">53 (24.2)</td><td align="left" valign="top">67 (30.6)</td><td align="left" valign="top">59 (26.9)</td><td align="left" valign="top">.10</td></tr><tr><td align="left" valign="top" colspan="6">OLA<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> (n=118 questions)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Success rate (%), mean (SD)</td><td align="left" valign="top">94 (10)</td><td align="left" valign="top">93 (12)</td><td align="left" valign="top">90 (14)</td><td align="left" valign="top">90 (14)</td><td align="left" valign="top">.008</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Success rates were compared using the Friedman test, and hallucination and confusion rates were compared using the Cochran <italic>Q</italic> test; test statistics are reported in the Results section.</p></fn><fn id="table1fn2"><p><sup>b</sup>IADE: Infirmier Anesth&#x00E9;siste Dipl&#x00F4;m&#x00E9; d&#x2019;&#x00C9;tat.</p></fn><fn id="table1fn3"><p><sup>c</sup>EDAIC: European Diploma in Anaesthesiology and Intensive Care.</p></fn><fn id="table1fn4"><p><sup>d</sup>OLA: On-Line Assessment.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Infirmier Anesth&#x00E9;siste Dipl&#x00F4;m&#x00E9; d'&#x00C9;tat examination success rates by subject domain. <italic>P</italic> values are Holm adjusted across the 5 domain-level Friedman tests.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Subjects</td><td align="left" valign="bottom">Claude Sonnet 4.5, mean (SD)</td><td align="left" valign="bottom">Gemini 2.5 Pro, mean (SD)</td><td align="left" valign="bottom">GPT-5, mean (SD)</td><td align="left" valign="bottom">Grok 4, mean (SD)</td><td align="left" valign="bottom">Adjusted <italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Procedure-related considerations (%; n=21)</td><td align="left" valign="top">87 (17)</td><td align="left" valign="top">90 (16)</td><td align="left" valign="top">89 (15)</td><td align="left" valign="top">86 (19)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top">Pharmacology (%; n=20)</td><td align="left" valign="top">92 (13)</td><td align="left" valign="top">92 (14)</td><td align="left" valign="top">90 (15)</td><td align="left" valign="top">89 (16)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top">Physiology (%; n=21)</td><td align="left" valign="top">99 (4)</td><td align="left" valign="top">93 (12)</td><td align="left" valign="top">93 (11)</td><td align="left" valign="top">95 (8)</td><td align="left" valign="top">.06</td></tr><tr><td align="left" valign="top">Techniques (%; n=14)</td><td align="left" valign="top">93 (9)</td><td align="left" valign="top">86 (12)</td><td align="left" valign="top">84 (11)</td><td align="left" valign="top">93 (9)</td><td align="left" valign="top">.16</td></tr><tr><td align="left" valign="top">Patient-specific factors (%; n=24)</td><td align="left" valign="top">93 (11)</td><td align="left" valign="top">93 (12)</td><td align="left" valign="top">94 (10)</td><td align="left" valign="top">95 (8)</td><td align="left" valign="top">&#x003E;.99</td></tr></tbody></table></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Pairwise comparisons of hallucination and confusion rates between large language models for Infirmier Anesth&#x00E9;siste Dipl&#x00F4;m&#x00E9; d'&#x00C9;tat questions: generalized estimating equation analysis.<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparisons</td><td align="left" valign="bottom">OR<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom">Adjusted <italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Hallucinations</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Pro vs Claude Sonnet 4.5</td><td align="left" valign="top">1.54 (0.84&#x2010;2.84)</td><td align="left" valign="top">.82</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5 vs Claude Sonnet 4.5</td><td align="left" valign="top">1.21 (0.63&#x2010;2.30)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs Claude Sonnet 4.5</td><td align="left" valign="top">1.54 (0.88&#x2010;2.70)</td><td align="left" valign="top">.78</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5 vs Gemini 2.5 Pro</td><td align="left" valign="top">0.78 (0.49&#x2010;1.26)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs Gemini 2.5 Pro</td><td align="left" valign="top">1.00 (0.63&#x2010;1.59)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs GPT-5</td><td align="left" valign="top">1.27 (0.79&#x2010;2.05)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top" colspan="3">Confusion</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Pro vs Claude Sonnet 4.5</td><td align="left" valign="top">1.72 (0.91&#x2010;3.24)</td><td align="left" valign="top">.34</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5 vs Claude Sonnet 4.5</td><td align="left" valign="top">1.87 (0.98&#x2010;3.57)</td><td align="left" valign="top">.34</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs Claude Sonnet 4.5</td><td align="left" valign="top">1.00 (0.59&#x2010;1.70)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5 vs Gemini 2.5 Pro</td><td align="left" valign="top">1.09 (0.70&#x2010;1.70)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs Gemini 2.5 Pro</td><td align="left" valign="top">0.58 (0.33&#x2010;1.02)</td><td align="left" valign="top">.34</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs GPT-5</td><td align="left" valign="top">0.53 (0.28&#x2010;1.02)</td><td align="left" valign="top">.34</td></tr></tbody></table> <table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Odds ratios were estimated using generalized estimating equations with a logit link, an exchangeable working correlation structure, and robust SEs, with questions specified as clusters; <italic>P</italic> values are Holm adjusted across the 6 pairwise comparisons within each outcome.</p></fn><fn id="table3fn2"><p><sup>b</sup>OR: odds ratio.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Pairwise comparisons of hallucination and confusion rates between large language models for European Diploma in Anaesthesiology and Intensive Care questions: generalized estimating equation analysis.<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparisons</td><td align="left" valign="bottom">OR<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom">Adjusted <italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Hallucinations</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Pro vs Claude Sonnet 4.5</td><td align="left" valign="top">0.84 (0.59&#x2010;1.20)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5 vs Claude Sonnet 4.5</td><td align="left" valign="top">1.14 (0.79&#x2010;1.63)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs Claude Sonnet 4.5</td><td align="left" valign="top">1.28 (0.91&#x2010;1.79)</td><td align="left" valign="top">.63</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5 vs Gemini 2.5 Pro</td><td align="left" valign="top">1.36 (0.99&#x2010;1.86)</td><td align="left" valign="top">.30</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs Gemini 2.5 Pro</td><td align="left" valign="top">1.52 (1.08&#x2010;2.14)</td><td align="left" valign="top">.09</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs GPT-5</td><td align="left" valign="top">1.13 (0.80&#x2010;1.58)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top" colspan="3">Confusion</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Pro vs Claude Sonnet 4.5</td><td align="left" valign="top">1.05 (0.76&#x2010;1.46)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5 vs Claude Sonnet 4.5</td><td align="left" valign="top">1.45 (1.03&#x2010;2.05)</td><td align="left" valign="top">.21</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs Claude Sonnet 4.5</td><td align="left" valign="top">1.21 (0.87&#x2010;1.70)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5 vs Gemini 2.5 Pro</td><td align="left" valign="top">1.38 (1.01&#x2010;1.88)</td><td align="left" valign="top">.21</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs Gemini 2.5 Pro</td><td align="left" valign="top">1.15 (0.86&#x2010;1.55)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok 4 vs GPT-5</td><td align="left" valign="top">0.84 (0.60-1.16)</td><td align="left" valign="top">&#x003E;.99</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Odds ratios were estimated using generalized estimating equations with a logit link, an exchangeable working correlation structure, and robust SEs, with questions specified as clusters; <italic>P</italic> values are Holm adjusted across the 6 pairwise comparisons within each outcome.</p></fn><fn id="table4fn2"><p><sup>b</sup>OR: odds ratio.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Success rates of 4 large language models across 3 anesthesiology examination types (Infirmier Anesth&#x00E9;siste Dipl&#x00F4;m&#x00E9; d'&#x00C9;tat [IADE]: n=100 questions; European Diploma in Anaesthesiology and Intensive Care [EDAIC]: n=219 questions; On-Line Assessment [OLA]: n=118 questions). Bars represent mean success rates; error bars denote 95% CIs of the mean. Brackets above each examination group indicate <italic>P</italic> values from the Friedman test for overall intermodel comparison.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e95859_fig01.png"/></fig></sec><sec id="s3-2"><title>IADE Examination Performance</title><p>IADE performance results are summarized in <xref ref-type="table" rid="table1">Tables 1</xref> and <xref ref-type="table" rid="table2">2</xref>. Overall performance was high across all models, with no statistically significant intermodel differences observed (Friedman <italic>&#x03C7;</italic><sup>2</sup><sub>3</sub>=5.2; <italic>P</italic>=.16; <italic>W</italic>=0.02).</p><p>Questions were categorized into 5 domains: physiology (21/100, 21%), pharmacology (20/100, 20%), anesthetic techniques and implementation (14/100, 14%), procedure-related considerations (21/100, 21%), and patient-specific factors (24/100, 24%). After Holm adjustment across the 5 domain-level Friedman tests, no significant intermodel differences were observed (physiology: <italic>&#x03C7;</italic><sup>2</sup><sub>3</sub>=11.1, adjusted <italic>P</italic>=.06, and <italic>W</italic>=0.18; anesthetic techniques: <italic>&#x03C7;</italic><sup>2</sup><sub>3</sub>=8.3, adjusted <italic>P</italic>=.16, and <italic>W</italic>=0.20; pharmacology, procedure-related considerations, and patient-specific factors: adjusted <italic>P</italic>&#x003E;.99). Given the small per-domain sample sizes, these analyses should be interpreted as exploratory.</p></sec><sec id="s3-3"><title>EDAIC Examination Performance</title><p>EDAIC performance results are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. Overall performance remained high across all models, with significant overall intermodel differences observed (Friedman <italic>&#x03C7;</italic><sup>2</sup><sub>3</sub>=13.9; <italic>P</italic>=.003; <italic>W</italic>=0.02). Claude Sonnet 4.5 and Gemini 2.5 Pro achieved the highest overall performances, whereas GPT-5 and Grok 4 showed lower mean scores.</p><p>In pairwise Wilcoxon signed-rank comparisons, Gemini 2.5 Pro significantly outperformed GPT-5 (<italic>Z</italic>=2.75; <italic>r</italic>=0.32; adjusted <italic>P</italic>=.04). No other pairwise comparisons were significant (adjusted <italic>P</italic>&#x2265;.09 in all cases).</p></sec><sec id="s3-4"><title>OLA Examination Performance</title><p>OLA performance results are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. Overall performance remained high across all models, with statistically significant overall intermodel differences observed (Friedman <italic>&#x03C7;</italic><sup>2</sup><sub>3</sub>=12.0; <italic>P</italic>=.008; <italic>W</italic>=0.03). Claude Sonnet 4.5 achieved the highest performance, whereas GPT-5 and Grok 4 showed lower mean scores. In pairwise Wilcoxon signed-rank comparisons, Claude Sonnet 4.5 significantly outperformed Grok 4 (<italic>Z</italic>=3.18; <italic>r</italic>=0.53; adjusted <italic>P</italic>=.009). No other pairwise comparisons were significant (adjusted <italic>P</italic>&#x2265;.08 in all cases).</p></sec><sec id="s3-5"><title>Hallucination and Confusion Analysis</title><p>Hallucination and confusion rates for each model are shown in <xref ref-type="table" rid="table1">Table 1</xref>. For IADE questions, hallucination rates did not differ significantly across models (Cochran <italic>Q</italic><sub>3</sub>=3.38; <italic>P</italic>=.34; <italic>W</italic>=0.01). The omnibus test for confusion rates reached nominal significance (Cochran <italic>Q</italic><sub>3</sub>=7.85; <italic>P</italic>=.049; <italic>W</italic>=0.03); however, the effect size was negligible, and no pairwise contrast was significant after Holm adjustment in the GEE analysis (<xref ref-type="table" rid="table3">Table 3</xref>; adjusted <italic>P</italic>&#x2265;.34 in all cases).</p><p>For EDAIC questions, neither hallucination rates (Cochran <italic>Q</italic><sub>3</sub>=6.22; <italic>P</italic>=.10; <italic>W</italic>=0.01) nor confusion rates (Cochran <italic>Q</italic><sub>3</sub>=6.28; <italic>P</italic>=.10; <italic>W</italic>=0.01) differed significantly across models. In the pairwise GEE analyses (<xref ref-type="table" rid="table3">Tables 3</xref> and <xref ref-type="table" rid="table4">4</xref>), no contrast was significant after Holm adjustment for either outcome; the largest contrast was observed for EDAIC hallucinations between Grok 4 and Gemini 2.5 Pro (odds ratio 1.52, 95% CI 1.08&#x2010;2.14; adjusted <italic>P</italic>=.09).</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study, we compared the performance of 4 current-generation LLMs across multiple European anesthesiology examination datasets and evaluated hallucination and question-related confusion rates. All evaluated models demonstrated high overall performance, with success rates approaching or exceeding typical human passing thresholds for the EDAIC examination. Although overall intermodel differences were detected, a single pairwise difference per examination remained significant after correction, indicating that no model was consistently superior. Importantly, hallucination rates remained nonnegligible across all models, highlighting an important limitation for unsupervised educational use. Notably, LLM performance surpassed typical first-attempt pass scores for the EDAIC, which range between 60% and 75% (2025 passing scores: 67% for part A and 73.5% for part B), indicating that LLMs may outperform average examinees on standardized theoretical examinations [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>The higher performance observed on IADE compared with EDAIC questions may reflect differences in cognitive complexity, as questions requiring lower-order cognitive skills (recall and comprehension) may be more amenable to current LLM capabilities than those requiring higher-order reasoning (analysis, synthesis, and evaluation) [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. However, because IADE questions were administered exclusively in French whereas EDAIC and OLA questions were administered exclusively in English, the effects of examination type and language cannot be disentangled. Consequently, the observed performance differences may reflect differences in cognitive complexity, language, or examination format or a combination of these factors.</p><p>After Holm-Bonferroni correction, a single robust pairwise difference remained per examination: Gemini outperformed GPT-5 on the EDAIC, and Claude Sonnet 4.5 outperformed Grok 4 on the OLA. However, absolute performance differences between models were modest (2&#x2010;4 percentage points), suggesting that, while statistically detectable, these disparities may have limited practical significance. The clinically relevant finding is that all models performed comparably well, and hallucination rates represent a more meaningful differentiator for educational deployment.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>These findings represent substantial improvement from previous reports. We evaluated GPT-3.5 in 2023 for the EDAIC part 1, finding a 70.5% success rate [<xref ref-type="bibr" rid="ref1">1</xref>]. Other authors have documented similar performance with earlier model versions [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. Our results demonstrate the rapid evolution of these technologies, consistent with other work showing progressive improvement across ChatGPT versions on medical licensing examinations worldwide [<xref ref-type="bibr" rid="ref20">20</xref>]. In the United States, a recent exploratory study showed that ChatGPT can perform similarly to humans in standardized oral examinations as graded by American Board of Anesthesiology examiners [<xref ref-type="bibr" rid="ref21">21</xref>], and other data show that current LLMs can surpass the minimum competency required for American Board of Anesthesiology certification [<xref ref-type="bibr" rid="ref22">22</xref>]. Remarkable data have also emerged from various medical fields in recent years suggesting that LLMs can achieve high levels of performance in knowledge-based tasks while exhibiting important limitations in complex clinical reasoning and reliability [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>Regarding hallucinations, our data do not support declaring any LLM superior. Confusion rates showed no significant intermodel differences except for a marginal effect on the IADE set (<italic>P</italic>=.049; <italic>W</italic>=0.03), which was of negligible magnitude and not corroborated by the pairwise GEE analysis. Training data may naturally contain errors that LLMs perpetuate, and users lack direct control over consulted sources, making verification challenging. Research on GPT-4o using Chilean national examination questions demonstrated that, despite strong factual recall and comprehension, the model exhibited limitations in complex reasoning and diagnostic judgment, producing unfounded medical assertions and imprecise conclusions [<xref ref-type="bibr" rid="ref11">11</xref>]. Importantly, the hallucination rates of 11% (11/100) to 20.1% (44/219) in our study carry significant practical implications for medical education. A student relying on an LLM for self-directed learning may encounter erroneous information in approximately 1 out of every 5 to 10 queries, potentially reinforcing incorrect medical knowledge if not critically evaluated. Unlike factual errors that may be easily identified, LLM hallucinations are typically presented with high confidence and syntactic fluency, making them particularly insidious in an educational context. This underscores the necessity for supervised use and systematic cross-referencing with authoritative sources when LLMs are used as learning aids.</p></sec><sec id="s4-3"><title>Implications for Education and Practice</title><p>From an educational perspective, these findings suggest that LLMs should not be used as primary knowledge sources but rather as guided learning assistants. A pragmatic framework for safe integration may include (1) mandatory cross-verification with authoritative resources, (2) supervised use in structured teaching environments, and (3) explicit training of students to recognize and critically appraise AI-generated content. Without such safeguards, the observed hallucination rates could lead to reinforcement of incorrect medical knowledge during self-directed learning.</p><p>Our standardized prompt approach, while ensuring reproducibility, may not reflect optimal performance achievable through advanced prompting techniques. Chain-of-thought prompting, few-shot learning, or retrieval-augmented generation could improve accuracy and reduce hallucinations. Future studies should systematically evaluate the impact of prompt engineering on medical examination performance.</p><p>The high acceptance of AI tools among medical students further supports the relevance of these findings [<xref ref-type="bibr" rid="ref25">25</xref>]. Surveys indicate widespread enthusiasm for AI integration into medical curricula, with students anticipating improvements in learning conditions and educational efficiency [<xref ref-type="bibr" rid="ref26">26</xref>]. LLMs enable personalized educational approaches through tutorial dialogue simulation, complex concept summarization, and enhanced knowledge retention, positioning them as valuable complementary tools in medical education [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Beyond student support, these tools may reduce educator administrative burden by automating repetitive tasks, allowing more time for personalized student guidance and improving pedagogical quality [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. The integration of LLMs into medical education raises important regulatory and policy considerations. While tools for self-directed learning may require less oversight than clinical decision support systems, educational institutions should develop clear policies regarding appropriate use, required supervision, and assessment of AI-generated content accuracy. Professional societies and accreditation bodies may need to establish guidelines for LLM use in medical education curricula.</p><p>Future studies examining open-ended questions or clinical cases would be valuable for testing LLM clinical reasoning capabilities. Given the exponential evolution of these technologies, our results will likely become obsolete within months, warranting repeated evaluation of subsequent model versions. Additionally, intervention studies evaluating the efficacy of LLM-assisted learning compared to traditional methods would help establish evidence-based recommendations for educational integration.</p></sec><sec id="s4-4"><title>Limitations</title><p>Several limitations warrant consideration. The consensus-based evaluation approach, while ensuring classification consistency between examiners, did not involve independent parallel scoring and, therefore, did not allow for formal quantification of interrater agreement through metrics such as the Cohen &#x03BA;. Future studies should incorporate independent blinded evaluations to enable formal reliability assessment. Question difficulty was not formally graded. The evaluation relied on MCQs, which primarily assess knowledge recall and pattern recognition rather than complex clinical reasoning.</p><p>Questions originated from official institutions evaluating future professionals (EDAIC and IADE), and dual review reduced evaluation bias. While our substantial question sample merits expansion for enhanced statistical reliability, the exclusive use of MCQs (rather than open-ended questions) attenuated evaluation bias. Prompt formulation was standardized, with each LLM receiving identical questions in new conversation threads to prevent cross-contamination, although prior model exposure to these question banks cannot be excluded.</p><p>An important limitation is the potential presence of some examination questions or closely related content within LLM training datasets (&#x201C;training data contamination&#x201D;). Consequently, high performance may partially reflect memorization or prior exposure rather than genuine reasoning ability. However, several factors mitigate this concern: IADE questions were administered in French from an institutional repository with limited online dissemination, EDAIC questions are regularly updated by the examination committee, and the use of 3 distinct sources with different formats and languages reduces the likelihood of systematic contamination across all question sets. This concern has been increasingly acknowledged in the LLM benchmarking literature: prior work has demonstrated that training data contamination can artificially inflate performance on widely used medical examination benchmarks such as MedQA and US Medical Licensing Examination&#x2013;style datasets, and dedicated detection methodologies have been proposed to estimate the extent of such contamination [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>].</p><p>Additional methodological limitations include the fact that (1) questions were administered in their original language (French for the IADE and English for the EDAIC and OLA) and potential translation or language processing effects were not explored; (2) exact model version checkpoints were not documented beyond the October 2025 time frame, limiting precise reproducibility; (3) no test-retest reliability assessment was performed to evaluate response consistency&#x2014;this single-query design was chosen to reflect typical real-world use patterns, where students or clinicians query an LLM once per question rather than repeatedly. However, future studies should assess response reproducibility across multiple runs.</p><p>Because the IADE questions were administered exclusively in French and the EDAIC and OLA questions were administered exclusively in English, language and examination type are inherently confounded in this design; consequently, the observed performance differences cannot be attributed to examination type alone and should be interpreted with this limitation in mind. Finally, there is a temporal limitation given the rapid LLM evolution as these results reflect October 2025 model versions.</p></sec><sec id="s4-5"><title>Conclusions</title><p>Current-generation LLMs demonstrated high performance on anesthesiology examination questions. All models substantially exceeded human passing thresholds, consistent with previously reported progressive improvements in successive LLM iterations. However, hallucination rates remained persistent across the 4 models, emphasizing the importance of user awareness regarding this phenomenon. AI-generated outputs continue to require critical verification and supervised use in educational settings, but these models represent tools with substantial pedagogical and assessment potential. These findings highlight both the educational potential and the limitations of LLMs as supplementary learning tools for medical trainees preparing for standardized examinations. Educational deployment frameworks must therefore prioritize systematic verification mechanisms over reliance on model accuracy alone.</p></sec></sec></body><back><ack><p>This manuscript was prepared with language editing assistance from ChatGPT (OpenAI) and Claude (Anthropic). These AI tools had no role in the study design, data acquisition, or interpretation of results. The final content was verified and approved by the authors.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The question corpus cannot be disclosed due to institutional confidentiality requirements. Statistical outputs are available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: SA, MS, AG</p><p>Data curation: SA, TG, AB</p><p>Formal analysis: SA</p><p>Methodology: SA, MS, AG</p><p>Validation: TG, AB, MS, AG</p><p>Writing&#x2014;original draft: SA, TG, AB, MS</p><p>Writing&#x2014;review and editing: MP, ST, PM, AG</p><p>All authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">EDAIC</term><def><p>European Diploma in Anaesthesiology and Intensive Care</p></def></def-item><def-item><term id="abb2">GEE</term><def><p>generalized estimating equations</p></def></def-item><def-item><term id="abb3">IADE</term><def><p>infirmier anesth&#x00E9;siste dipl&#x00F4;m&#x00E9; d&#x2019;&#x00E9;tat</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">MCQ</term><def><p>multiple-choice question</p></def></def-item><def-item><term id="abb6">OLA</term><def><p>On-Line Assessment</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Andrei</surname><given-names>S</given-names> </name><name name-style="western"><surname>Longrois</surname><given-names>D</given-names> </name><name name-style="western"><surname>Stefan</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Stefan</surname><given-names>G</given-names> </name></person-group><article-title>Chat-based generative pretrained transformers sits the European Diploma in Anaesthesiology and Intensive Care part I examination: a computational study</article-title><source>Eur J Anaesthesiol</source><year>2024</year><month>04</month><day>1</day><volume>41</volume><issue>4</issue><fpage>323</fpage><lpage>325</lpage><pub-id pub-id-type="doi">10.1097/EJA.0000000000001969</pub-id><pub-id pub-id-type="medline">38426256</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Benito</surname><given-names>P</given-names> </name><name name-style="western"><surname>Isla-Jover</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gonz&#x00E1;lez-Castro</surname><given-names>P</given-names> </name><etal/></person-group><article-title>GPT-4o and OpenAI o1 performance on the 2024 Spanish competitive medical specialty access examination: cross-sectional quantitative evaluation study</article-title><source>JMIR Med Educ</source><year>2026</year><month>01</month><day>12</day><volume>12</volume><fpage>e75452</fpage><pub-id pub-id-type="doi">10.2196/75452</pub-id><pub-id pub-id-type="medline">41525685</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gilson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Safranek</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>How does ChatGPT perform on the United States Medical Licensing Examination (USMLE)? The implications of large language models for medical education and knowledge assessment</article-title><source>JMIR Med Educ</source><year>2023</year><month>02</month><day>8</day><volume>9</volume><fpage>e45312</fpage><pub-id pub-id-type="doi">10.2196/45312</pub-id><pub-id pub-id-type="medline">36753318</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Siam</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Varela</surname><given-names>A</given-names> </name><name name-style="western"><surname>Faruk</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>Benchmarking large language models on the United States Medical Licensing Examination for clinical reasoning and medical licensing scenarios</article-title><source>Sci Rep</source><year>2025</year><month>12</month><day>3</day><volume>16</volume><issue>1</issue><fpage>1387</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-31010-4</pub-id><pub-id pub-id-type="medline">41339739</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferretti</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rizzi</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Requena</surname><given-names>LS</given-names> </name><name name-style="western"><surname>Bicudo</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Hamamoto Filho</surname><given-names>PT</given-names> </name></person-group><article-title>Factors associated with lower performance of artificial intelligence on answering undergraduate medical education multiple-choice questions</article-title><source>Med Sci Educ</source><year>2025</year><volume>35</volume><issue>4</issue><fpage>2145</fpage><lpage>2152</lpage><pub-id pub-id-type="doi">10.1007/s40670-025-02426-4</pub-id><pub-id pub-id-type="medline">41112890</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ravi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Neinstein</surname><given-names>A</given-names> </name><name name-style="western"><surname>Murray</surname><given-names>SG</given-names> </name></person-group><article-title>Large language models and medical education: preparing for a rapid transformation in how trainees will learn to be doctors</article-title><source>ATS Sch</source><year>2023</year><volume>4</volume><issue>3</issue><fpage>282</fpage><lpage>292</lpage><pub-id pub-id-type="doi">10.34197/ats-scholar.2023-0036PS</pub-id><pub-id pub-id-type="medline">37795112</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Anh-Hoang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tran</surname><given-names>V</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>LM</given-names> </name></person-group><article-title>Survey and analysis of hallucinations in large language models: attribution to prompting strategies or model behavior</article-title><source>Front Artif Intell</source><year>2025</year><volume>8</volume><fpage>1622292</fpage><pub-id pub-id-type="doi">10.3389/frai.2025.1622292</pub-id><pub-id pub-id-type="medline">41098969</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Metheniti</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bhar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Asher</surname><given-names>N</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bechet</surname><given-names>F</given-names> </name><name name-style="western"><surname>Chifu</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Pinel-Sauvagnat</surname><given-names>K</given-names> </name><name name-style="western"><surname>Favre</surname><given-names>B</given-names> </name><name name-style="western"><surname>Maes</surname><given-names>E</given-names> </name><name name-style="western"><surname>Nurbakova</surname><given-names>D</given-names> </name></person-group><article-title>Une revue sur les hallucinations des LLM</article-title><source>Actes des 32&#x00E8;me Conf&#x00E9;rence sur le Traitement Automatique des Langues Naturelles (TALN)</source><year>2025</year><access-date>2026-08-13</access-date><publisher-name>ATALA &#x0026; ARIA</publisher-name><fpage>756</fpage><lpage>779</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.jeptalnrecital-taln.46/">https://aclanthology.org/2025.jeptalnrecital-taln.46/</ext-link></comment></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Battaglia</surname><given-names>F</given-names> </name><name name-style="western"><surname>Mansuria</surname><given-names>K</given-names> </name><name name-style="western"><surname>Heyman</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Terlecky</surname><given-names>SR</given-names> </name></person-group><article-title>Beyond text generation: assessing large language models&#x2019; ability to reason logically and follow strict rules</article-title><source>AI</source><year>2025</year><volume>6</volume><issue>1</issue><fpage>12</fpage><pub-id pub-id-type="doi">10.3390/ai6010012</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>W</given-names> </name><etal/></person-group><article-title>A survey on hallucination in large language models: principles, taxonomy, challenges, and open questions</article-title><source>ACM Trans Inf Syst</source><year>2025</year><volume>43</volume><issue>2</issue><fpage>1</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.1145/3703155</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Altermatt</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Neyem</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sumonte</surname><given-names>NI</given-names> </name><etal/></person-group><article-title>Evaluating GPT-4o in high-stakes medical assessments: performance and error analysis on a Chilean anesthesiology exam</article-title><source>BMC Med Educ</source><year>2025</year><month>10</month><day>27</day><volume>25</volume><issue>1</issue><fpage>1499</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-08084-9</pub-id><pub-id pub-id-type="medline">41146119</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="medline">37438534</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ateleanu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Goldik</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Varvinskiy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Moisin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Hasan</surname><given-names>N</given-names> </name></person-group><article-title>An overview of the European Diploma of Anaesthesia and Intensive Care and of other important initiatives of the European Society of Anaesthesiology</article-title><source>Jurnalul Rom&#x00E2;n de Anestezie Terapie Intensiv&#x0103;</source><year>2013</year><access-date>2026-08-18</access-date><volume>20</volume><issue>2</issue><fpage>137</fpage><lpage>144</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.researchgate.net/publication/288566101_An_overview_of_the_European_Diploma_of_Anaesthesia_and_Intensive_Care_and_of_other_important_initiatives_of_the_European_Society_of_Anaesthesiology">https://www.researchgate.net/publication/288566101_An_overview_of_the_European_Diploma_of_Anaesthesia_and_Intensive_Care_and_of_other_important_initiatives_of_the_European_Society_of_Anaesthesiology</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brogly</surname><given-names>N</given-names> </name><name name-style="western"><surname>Varvinskiy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Varosyan</surname><given-names>A</given-names> </name><etal/></person-group><article-title>On-Line Assessment (OLA) as a preparation for the European Diploma in Anaesthesiology and Intensive Care (EDAIC), a retrospective observational study on the results and the potential impact on the part-I examination</article-title><source>Rev Esp Anestesiol Reanim (Engl Ed)</source><year>2022</year><month>10</month><volume>69</volume><issue>8</issue><fpage>454</fpage><lpage>462</lpage><pub-id pub-id-type="doi">10.1016/j.redare.2022.08.006</pub-id><pub-id pub-id-type="medline">36089526</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>LOI n&#x00B0; 2012-300 du 5 mars 2012 relative aux recherches impliquant la personne humaine (1)</article-title><source>R&#x00E9;publique Fran&#x00E7;aise</source><access-date>2026-08-18</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.legifrance.gouv.fr/loda/id/JORFTEXT000025441587/">https://www.legifrance.gouv.fr/loda/id/JORFTEXT000025441587/</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bubeck</surname><given-names>S</given-names> </name><name name-style="western"><surname>Petro</surname><given-names>J</given-names> </name></person-group><article-title>Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine</article-title><source>N Engl J Med</source><year>2023</year><month>03</month><day>30</day><volume>388</volume><issue>13</issue><fpage>1233</fpage><lpage>1239</lpage><pub-id pub-id-type="doi">10.1056/NEJMsr2214184</pub-id><pub-id pub-id-type="medline">36988602</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lucas</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Upperman</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>JR</given-names> </name></person-group><article-title>A systematic review of large language models and their implications in medical education</article-title><source>Med Educ</source><year>2024</year><month>11</month><volume>58</volume><issue>11</issue><fpage>1276</fpage><lpage>1285</lpage><pub-id pub-id-type="doi">10.1111/medu.15402</pub-id><pub-id pub-id-type="medline">38639098</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kung</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Cheatham</surname><given-names>M</given-names> </name><name name-style="western"><surname>Medenilla</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title><source>PLOS Digit Health</source><year>2023</year><month>02</month><volume>2</volume><issue>2</issue><fpage>e0000198</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id><pub-id pub-id-type="medline">36812645</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yoon</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Oh</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>BG</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>HJ</given-names> </name></person-group><article-title>Performance of ChatGPT in the in-training examination for anesthesiology and pain medicine residents in South Korea: observational study</article-title><source>JMIR Med Educ</source><year>2024</year><month>09</month><day>16</day><volume>10</volume><fpage>e56859</fpage><pub-id pub-id-type="doi">10.2196/56859</pub-id><pub-id pub-id-type="medline">39284182</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khakpaki</surname><given-names>A</given-names> </name></person-group><article-title>Advancements in artificial intelligence transforming medical education: a comprehensive overview</article-title><source>Med Educ Online</source><year>2025</year><month>12</month><volume>30</volume><issue>1</issue><fpage>2542807</fpage><pub-id pub-id-type="doi">10.1080/10872981.2025.2542807</pub-id><pub-id pub-id-type="medline">40798935</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blacker</surname><given-names>SN</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>F</given-names> </name><name name-style="western"><surname>Winecoff</surname><given-names>D</given-names> </name><etal/></person-group><article-title>An exploratory analysis of ChatGPT compared to human performance with the anesthesiology oral board examination: initial insights and implications</article-title><source>Anesth Analg</source><year>2025</year><month>06</month><day>1</day><volume>140</volume><issue>6</issue><fpage>1253</fpage><lpage>1262</lpage><pub-id pub-id-type="doi">10.1213/ANE.0000000000006875</pub-id><pub-id pub-id-type="medline">39269908</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ngo</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wilhelmi</surname><given-names>B</given-names> </name></person-group><article-title>Evaluating large language models on American Board of Anesthesiology-style anesthesiology questions: accuracy, domain consistency, and clinical implications</article-title><source>J Cardiothorac Vasc Anesth</source><year>2025</year><month>09</month><volume>39</volume><issue>9</issue><fpage>2511</fpage><lpage>2515</lpage><pub-id pub-id-type="doi">10.1053/j.jvca.2025.05.033</pub-id><pub-id pub-id-type="medline">40518333</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chan-Chia Lin</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Zwei-Chieng Chang</surname><given-names>J</given-names> </name></person-group><article-title>Performance of artificial intelligence chatbots in national dental licensing examination</article-title><source>J Dent Sci</source><year>2025</year><month>10</month><volume>20</volume><issue>4</issue><fpage>2307</fpage><lpage>2314</lpage><pub-id pub-id-type="doi">10.1016/j.jds.2025.05.012</pub-id><pub-id pub-id-type="medline">41040620</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Genovese</surname><given-names>A</given-names> </name><name name-style="western"><surname>Borna</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gomez-Cabello</surname><given-names>CA</given-names> </name><etal/></person-group><article-title>The current landscape of artificial intelligence in plastic surgery education and training: a systematic review</article-title><source>J Surg Educ</source><year>2025</year><month>08</month><volume>82</volume><issue>8</issue><fpage>103519</fpage><pub-id pub-id-type="doi">10.1016/j.jsurg.2025.103519</pub-id><pub-id pub-id-type="medline">40378641</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>S</given-names> </name></person-group><article-title>Medical undergraduate students&#x2019; readiness and anxiety toward artificial intelligence: a systematic review and meta-analysis</article-title><source>BMC Med Educ</source><year>2025</year><month>12</month><day>4</day><volume>26</volume><issue>1</issue><fpage>44</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-08388-w</pub-id><pub-id pub-id-type="medline">41339885</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chakri</surname><given-names>I</given-names> </name><name name-style="western"><surname>El Khayali</surname><given-names>O</given-names> </name><name name-style="western"><surname>Lahlou</surname><given-names>L</given-names> </name></person-group><article-title>Knowledge and perceptions of AI among medical students in Morocco: cross-sectional study</article-title><source>JMIR Form Res</source><year>2025</year><month>09</month><day>19</day><volume>9</volume><fpage>e66156</fpage><pub-id pub-id-type="doi">10.2196/66156</pub-id><pub-id pub-id-type="medline">40971792</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>F</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Large language models and medical education: a paradigm shift in educator roles</article-title><source>Smart Learn Environ</source><year>2024</year><volume>11</volume><fpage>26</fpage><pub-id pub-id-type="doi">10.1186/s40561-024-00313-w</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Roveta</surname><given-names>A</given-names> </name><name name-style="western"><surname>Castello</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Massarino</surname><given-names>C</given-names> </name><name name-style="western"><surname>Francese</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ugo</surname><given-names>F</given-names> </name><name name-style="western"><surname>Maconi</surname><given-names>A</given-names> </name></person-group><article-title>Artificial intelligence in medical education: a narrative review on implementation, evaluation, and methodological challenges</article-title><source>AI</source><year>2025</year><volume>6</volume><issue>9</issue><fpage>227</fpage><pub-id pub-id-type="doi">10.3390/ai6090227</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Jacovi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Caciularu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Goldman</surname><given-names>O</given-names> </name><name name-style="western"><surname>Goldberg</surname><given-names>Y</given-names> </name></person-group><article-title>Stop uploading test data in plain text: practical strategies for mitigating data contamination by evaluation benchmarks</article-title><source>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>5075</fpage><lpage>5084</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.308</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sainz</surname><given-names>O</given-names> </name><name name-style="western"><surname>Campos</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Garc&#x00ED;a-Ferrero</surname><given-names>I</given-names> </name><name name-style="western"><surname>Etxaniz</surname><given-names>J</given-names> </name><name name-style="western"><surname>de Lacalle</surname><given-names>OL</given-names> </name><name name-style="western"><surname>Agirre</surname><given-names>E</given-names> </name></person-group><article-title>NLP evaluation in trouble: on the need to measure LLM data contamination for each benchmark</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 27, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2310.18018</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Representative examples of hallucination and question-related confusion classifications.</p><media xlink:href="formative_v10i1e95859_app1.docx" xlink:title="DOCX File, 20 KB"/></supplementary-material></app-group></back></article>