<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.0" xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JFR</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id>
      <journal-title>JMIR Formative Research</journal-title>
      <issn pub-type="epub">2561-326X</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v10i1e100148</article-id>
      <article-id pub-id-type="pmid"/>
      <article-id pub-id-type="doi">10.2196/100148</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Performance of ChatGPT, Claude, and AMBOSS on the European Board of Urology In-Service Assessment and Alignment With the European Association of Urology 2025 Guidelines: Comparative Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Sarvestan</surname>
            <given-names>Javad</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Cheng</surname>
            <given-names>Pai-Yu</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Eldaneen</surname>
            <given-names>Mohamed</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0008-6697-6053</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Hendy</surname>
            <given-names>Shaza</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-4898-4422</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Ramadan</surname>
            <given-names>Omar</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0004-0155-1629</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Nikolinakos</surname>
            <given-names>Panagiotis</given-names>
          </name>
          <degrees>MSc, MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-8427-7223</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Muneer</surname>
            <given-names>Asif</given-names>
          </name>
          <degrees>MBChB, MD, PhD, MBA</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <xref rid="aff4" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-2958-1614</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Alnajjar</surname>
            <given-names>Hussain M</given-names>
          </name>
          <degrees>BSc, MBBS, MCh</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <xref rid="aff4" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-6364-0310</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Pang</surname>
            <given-names>Karl H</given-names>
          </name>
          <degrees>BSc, MBChB, MSc, PhD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <address>
            <institution>Division of Surgery and Interventional Science</institution>
            <institution>University College London</institution>
            <addr-line>Gower Street</addr-line>
            <addr-line>London, WC1E 6BT, WC1E 6BT</addr-line>
            <country>United Kingdom</country>
            <phone>44 20 7679 2000</phone>
            <email>karlpang@doctors.org.uk</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-5703-2319</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Department of Urology, Chelsea and Westminster Hospital NHS Foundation Trust</institution>
        <addr-line>London</addr-line>
        <country>United Kingdom</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>East Kent Hospitals University NHS Foundation Trust</institution>
        <addr-line>Kent</addr-line>
        <country>United Kingdom</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>Division of Surgery and Interventional Science</institution>
        <institution>University College London</institution>
        <addr-line>London, WC1E 6BT</addr-line>
        <country>United Kingdom</country>
      </aff>
      <aff id="aff4">
        <label>4</label>
        <institution>University College London Hospitals NHS Foundation Trust</institution>
        <addr-line>London, England</addr-line>
        <country>United Kingdom</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Karl H Pang <email>karlpang@doctors.org.uk</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>2</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>10</volume>
      <elocation-id>e100148</elocation-id>
      <history>
        <date date-type="received">
          <day>3</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>1</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>2</day>
          <month>7</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>3</day>
          <month>7</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Mohamed Eldaneen, Shaza Hendy, Omar Ramadan, Panagiotis Nikolinakos, Asif Muneer, Hussain M Alnajjar, Karl H Pang. Originally published in JMIR Formative Research (https://formative.jmir.org), 02.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on https://formative.jmir.org, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://formative.jmir.org/2026/1/e100148" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Recent advances in AI, particularly large language models, have generated growing interest in their application to medical education and examination preparation. However, the accuracy, reasoning quality, and adherence to clinical guidelines of these tools in postgraduate urology assessments remain unclear.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to evaluate the performance of 3 AI tools, ChatGPT (GPT-4.0), Claude (version 4.5), and AMBOSS, on European Board of Urology (EBU)–style multiple-choice questions, with a particular focus on accuracy, insight, concordance, and adherence to European Association of Urology (EAU) guidelines.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>A total of 200 single-best-answer questions from the EBU In-Service Assessment workbook (2021-2022) were input into each AI model. Models were prompted to select an answer and provide an explanation. Two urologists with post–Fellowship of the Royal College of Surgeons (FRCS) training independently assessed the outputs. Accuracy was defined as correct answer selection. Concordance was defined as the logical alignment between the answer and its explanation. Insight was evaluated across 3 domains—nonobvious deduction, discriminative reasoning, and clinical validity—and was graded as low, moderate, or high.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>ChatGPT demonstrated the highest accuracy (171/200, 85.5%), compared to Claude and AMBOSS (both 159/200, 79.5%; <italic>P</italic>=.14). Concordance was also significantly higher for ChatGPT (190/200, 95%) than for Claude (176/200, 88%) and AMBOSS (152/200, 76%; <italic>P</italic>&lt;.001). Nonobvious deduction was predominantly low to moderate across all models, reflecting the recall-based nature of many questions. ChatGPT and Claude showed stronger discriminative reasoning, while AMBOSS demonstrated limited exclusion of alternative options. Clinical validity was high overall, with ChatGPT showing the greatest consistency with EAU guidelines. There was substantial agreement between the 2 reviewers (weighted κ coefficient &gt;0.61).</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>AI tools can achieve high accuracy on EBU-style assessments; however, differences in reasoning quality and guideline adherence are evident. ChatGPT demonstrated superior performance across all evaluated domains, supporting its role as a potential adjunct in postgraduate urology education.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>European Board of Urology</kwd>
        <kwd>EBU</kwd>
        <kwd>artificial intelligence</kwd>
        <kwd>AI</kwd>
        <kwd>ChatGPT</kwd>
        <kwd>Claude</kwd>
        <kwd>AMBOSS</kwd>
        <kwd>European Association of Urology guidelines</kwd>
        <kwd>EAU guidelines</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>The European Board of Urology (EBU) examination is a high-stakes assessment designed to evaluate core and advanced urological knowledge in trainees approaching completion of specialist training. Success in the EBU examination is often viewed as an indicator of readiness for independent practice and is closely aligned with the knowledge base required for fellowship-level examinations, such as the Fellowship of the Royal College of Surgeons (FRCS) exam [<xref ref-type="bibr" rid="ref1">1</xref>].</p>
      <p>The EBU examination is a 2-part assessment comprising a written theory examination (part 1) and an oral viva examination (part 2). The part 1 written examination consists of 110 single-best-answer multiple-choice questions (MCQs) designed to determine whether candidates meet the minimum knowledge standard defined by the EBU. The examination covers the breadth of urological practice and is structured across core domains, including basic science, oncology, endourology, andrology, functional urology, trauma, and other subspecialty areas. The part 2 examination is an oral viva that assesses clinical reasoning and decision-making through structured case-based discussions [<xref ref-type="bibr" rid="ref1">1</xref>].</p>
      <p>Advances in AI, particularly the development of large language models (LLMs), have generated substantial interest in medical education [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. Despite this promise, the application of LLMs in health care education remains contentious due to concerns regarding factual accuracy, hallucinated outputs, lack of transparency in reasoning, and the risk of overreliance by learners [<xref ref-type="bibr" rid="ref4">4</xref>]. Although several studies have demonstrated strong LLM performance in general medical and nonmedical professional examinations, including the United States Medical Licensing Examination (USMLE) and legal board assessments [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>], performance within specialty-specific, higher-order clinical domains, particularly those requiring nuanced decision-making, remains less well characterized [<xref ref-type="bibr" rid="ref8">8</xref>]. Recent work has suggested that AI systems enhanced with specialty-specific guidelines can achieve high performance on urology board-style questions, highlighting both the potential and limitations of such tools in specialist education [<xref ref-type="bibr" rid="ref9">9</xref>].</p>
      <p>In this study, we aimed to evaluate the performance of 3 widely used AI tools: 2 general-purpose models, ChatGPT (GPT-4.0) and Claude (version 4.5), and 1 medical-specific platform, AMBOSS, on a set of EBU-style urology MCQs. Unlike general LLMs, AMBOSS’s content and explanations are curated by medical educators and clinicians, which may enhance factual accuracy and guideline alignment in domain-specific settings [<xref ref-type="bibr" rid="ref10">10</xref>].</p>
      <p>In addition, model outputs were assessed for insight and concordance with expert reasoning, using benchmark comparisons against the European Association of Urology (EAU) guidelines [<xref ref-type="bibr" rid="ref11">11</xref>].</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <p>This study was performed with reference to the METRICS (Modern Quality Assessment Framework for Evaluating Generative AI Studies in Healthcare) framework [<xref ref-type="bibr" rid="ref12">12</xref>] (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
      <sec>
        <title>Study Design</title>
        <p>This was a comparative evaluation of 3 AI-based tools, ChatGPT (GPT-4.0; OpenAI), Claude (version 4.5; Anthropic PBC), and AMBOSS (AMBOSS GmbH), on EBU-style MCQs, benchmarked against expert clinician assessment using the 2025 EAU guidelines [<xref ref-type="bibr" rid="ref8">8</xref>] as a reference.</p>
        <p>Data collection was performed between December 8 and 12, 2025. ChatGPT was accessed using a ChatGPT Plus subscription through the web interface [<xref ref-type="bibr" rid="ref13">13</xref>], Claude through the Claude-4.5 Pro web interface [<xref ref-type="bibr" rid="ref14">14</xref>], and AMBOSS through a full-access institutional subscription [<xref ref-type="bibr" rid="ref15">15</xref>].</p>
        <p>Questions were entered manually using an identical prompt for all models: “You are sitting the European Board of Urology (EBU) written examination. For the following question, select the single best answer from the options provided. Provide your chosen answer and a brief explanation for your reasoning.”</p>
        <p>Each question was submitted once in a new, empty conversation to minimize context carryover between responses. No system prompts, custom instructions, or user-modifiable parameters were used. Output variability across repeated runs was not assessed and represents a limitation of the study.</p>
      </sec>
      <sec>
        <title>Question Source</title>
        <p>The most recent available EBU In-Service Assessment Questions booklet (2021-2022) was requested from the EBU by email, and approval was obtained for its use. This official workbook contains 200 single-best-answer MCQs with predetermined correct answers. The In-Service Assessment and its accompanying workbook are not official past papers of the EBU part 1 written examination but serve as structured preparatory and educational resources reflecting the style and scope of the certification examination and do not operate on a pass or fail basis.</p>
        <p>EBU examination outcomes are reportedly based on the mean score and SD of the candidate cohort, although official pass rates are not publicly available. Furthermore, the workbook is an educational resource. Therefore, performance in this study cannot be directly equated with performance on the official EBU part 1 examination.</p>
      </sec>
      <sec>
        <title>Assessment</title>
        <p>Accuracy was defined as whether the selected answer correctly addressed the question (correct vs incorrect).</p>
        <p>Concordance was defined as the logical alignment between the selected answer and its accompanying explanation, consistent with previously described LLM evaluation frameworks. Concordance was assessed on a per-question basis, with each of the 200 questions scored as either “yes” (in agreement with the expected answer) or “no.” The concordance percentage was then calculated.</p>
        <p>Insight assessment was adapted from prior studies evaluating the presence of nonobvious, clinically meaningful reasoning in AI-generated responses and was assessed across three domains: (1) nonobvious deduction (reasoning beyond the question stem), (2) discriminative reasoning (active exclusion of alternative options), and (3) clinical validity (factual accuracy and alignment with accepted urological practice and the EAU guidelines). Insight was graded as low (minimal or absent reasoning), moderate (partial reasoning), or high (clear, structured, and multistep reasoning). A 3-tier classification was selected to capture gradations in reasoning quality and explanatory depth that may not be adequately represented by a binary assessment [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>].</p>
        <p>Ratings were independently assigned by 2 assessors (2 post-FRCS urologists, ME and OR), and disagreements were resolved through consensus discussion, with adjudication by a senior reviewer (KHP) where required, consistent with established approaches for expert-rated assessment of complex constructs [<xref ref-type="bibr" rid="ref18">18</xref>].</p>
      </sec>
      <sec>
        <title>Analysis</title>
        <p>Descriptive data were presented as numbers and percentages with 95% CI, where appropriate. Paired categorical data were analyzed using the McNemar test for comparisons between 2 related groups and the Cochran <italic>Q</italic> test for comparisons among 3 or more related groups. Statistical significance was defined as <italic>P</italic>&lt;.05. The weighted κ coefficient was used to assess interrater reliability for agreement on concordance and insight categorical data [<xref ref-type="bibr" rid="ref19">19</xref>]. Results were tabulated, and graphs were plotted using Microsoft Excel (version 16).</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <p>The accuracy and concordance performance of the 3 AI models with corresponding 95% CIs are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. ChatGPT achieved the highest accuracy at 85.5% (171/200, 95% CI 80.6%-90.4%), compared with 79.5% (159/200, 95% CI 73.9%-85.1%) for both Claude and AMBOSS (<italic>P</italic>=.14). All 3 models achieved accuracy scores exceeding 70%, indicating a high level of performance on EBU-style questions.</p>
      <table-wrap position="float" id="table1">
        <label>Table 1</label>
        <caption>
          <p>Accuracy and concordance of ChatGPT, Claude, and AMBOSS, with 95% CIs (N=200).</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="30"/>
          <col width="350"/>
          <col width="0"/>
          <col width="250"/>
          <col width="0"/>
          <col width="200"/>
          <col width="0"/>
          <col width="170"/>
          <thead>
            <tr valign="top">
              <td colspan="3">Outcomes and models</td>
              <td colspan="2">Positive responses, n (%; 95% CI)</td>
              <td colspan="2">Cochran Q</td>
              <td><italic>P</italic> value</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td colspan="5">
                <bold>Accuracy</bold>
              </td>
              <td colspan="2">3.9</td>
              <td>.14<sup>a</sup></td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>ChatGPT</td>
              <td colspan="2">171 (85.5; 80.6-90.4)</td>
              <td colspan="2">
                <break/>
              </td>
              <td colspan="2">
                <break/>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>Claude</td>
              <td colspan="2">159 (79.5; 73.9-85.1)</td>
              <td colspan="2">
                <break/>
              </td>
              <td colspan="2">
                <break/>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>AMBOSS</td>
              <td colspan="2">159 (79.5; 73.9-85.1)</td>
              <td colspan="2">
                <break/>
              </td>
              <td colspan="2">
                <break/>
              </td>
            </tr>
            <tr valign="top">
              <td colspan="5">
                <bold>Concordance</bold>
              </td>
              <td colspan="2">29.3</td>
              <td>
                <italic>&lt;.001</italic>
                <sup>a,b</sup>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>ChatGPT</td>
              <td colspan="2">190 (95; 92.0-98.0)</td>
              <td colspan="2">
                <break/>
              </td>
              <td colspan="2">
                <italic>.02</italic>
                <sup>c</sup>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>Claude</td>
              <td colspan="2">176 (88; 83.5-92.5)</td>
              <td colspan="2">
                <break/>
              </td>
              <td colspan="2">
                <italic>&lt;.001</italic>
                <sup>b,d</sup>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>AMBOSS</td>
              <td colspan="2">152 (76; 70.1-81.9)</td>
              <td colspan="2">
                <break/>
              </td>
              <td colspan="2">
                <italic>.004</italic>
                <sup>b,e</sup>
              </td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table1fn1">
            <p><sup>a</sup>Comparison between the 3 AI tools (Cochran Q test).</p>
          </fn>
          <fn id="table1fn2">
            <p><sup>b</sup>Statistically significant (<italic>P</italic>&lt;.05).</p>
          </fn>
          <fn id="table1fn3">
            <p><sup>c</sup>Comparison between ChatGPT and Claude (McNemar).</p>
          </fn>
          <fn id="table1fn4">
            <p><sup>d</sup>Comparison between ChatGPT and AMBOSS.</p>
          </fn>
          <fn id="table1fn5">
            <p><sup>e</sup>Comparison between Claude and AMBOSS.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
      <p>ChatGPT significantly demonstrated the highest concordance at 95% (190/200, 95% CI 92.0%-98.0%), followed by Claude at 88% (176/200, 95% CI 83.5%-92.5%) and AMBOSS at 76% (152/200, 95% CI 70.1%-81.9%; <italic>P</italic>&lt;.001). There were also statistically significant differences observed between the different AI tools (<xref ref-type="table" rid="table1">Table 1</xref>). The weighted κ coefficient was 1.0, representing perfect agreement.</p>
      <p>When analyzing subspecialty topics, all 3 models achieved an accuracy of at least 68.1% across all domains (<xref ref-type="table" rid="table2">Table 2</xref>). An accuracy of 100% was achieved by all models in “miscellaneous” and “transplant and nephrology.” ChatGPT and Claude both achieved 100% accuracy in “trauma or emergency.” ChatGPT achieved the highest accuracy in “oncology” (82.6%), “pediatric and congenital urology” (90.9%), and “surgical principles” (100%). Claude performed best in “functional urology and benign prostatic hyperplasia” (85.7%) and “lithiasis and infections” (84.6%), while AMBOSS achieved the highest accuracy in “andrology and infertility” (93.8%).</p>
      <table-wrap position="float" id="table2">
        <label>Table 2</label>
        <caption>
          <p>Accuracy of ChatGPT, Claude, and AMBOSS across European Board of Urology subspecialty domains.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="500"/>
          <col width="170"/>
          <col width="150"/>
          <col width="180"/>
          <thead>
            <tr valign="top">
              <td>Domains</td>
              <td>ChatGPT, accuracy (%)</td>
              <td>Claude, accuracy (%)</td>
              <td>AMBOSS, accuracy (%)</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td>Andrology and infertility</td>
              <td>81.3</td>
              <td>81.3</td>
              <td>93.8</td>
            </tr>
            <tr valign="top">
              <td>Functional urology and benign prostatic hyperplasia</td>
              <td>75.0</td>
              <td>85.7</td>
              <td>75.0</td>
            </tr>
            <tr valign="top">
              <td>Lithiasis and infections</td>
              <td>80.8</td>
              <td>84.6</td>
              <td>80.8</td>
            </tr>
            <tr valign="top">
              <td>Miscellaneous</td>
              <td>100</td>
              <td>100</td>
              <td>100</td>
            </tr>
            <tr valign="top">
              <td>Oncology</td>
              <td>82.6</td>
              <td>68.1</td>
              <td>76.8</td>
            </tr>
            <tr valign="top">
              <td>Pediatric and congenital urology</td>
              <td>90.9</td>
              <td>81.8</td>
              <td>68.2</td>
            </tr>
            <tr valign="top">
              <td>Surgical principles</td>
              <td>100</td>
              <td>86.7</td>
              <td>80.0</td>
            </tr>
            <tr valign="top">
              <td>Transplant and nephrology</td>
              <td>100</td>
              <td>100</td>
              <td>100</td>
            </tr>
            <tr valign="top">
              <td>Trauma or emergency</td>
              <td>100</td>
              <td>100</td>
              <td>75.0</td>
            </tr>
          </tbody>
        </table>
      </table-wrap>
      <p>The results from the insight assessment are demonstrated in <xref rid="figure1" ref-type="fig">Figure 1</xref> and <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. For nonobvious deduction, ChatGPT demonstrated low, moderate, and high ratings in 47% (94/200), 48.5% (97/200), and 4.5% (9/200) of responses, respectively. Claude showed a similar distribution, with low, moderate, and high ratings in 48% (96/200), 49% (98/200), and 3% (6/200) of responses, respectively. AMBOSS demonstrated the lowest performance in this domain, with low, moderate, and high ratings in 60% (120/200), 39% (78/200), and 1% (2/200) of responses, respectively (<xref rid="figure1" ref-type="fig">Figure 1</xref>A).</p>
      <fig id="figure1" position="float">
        <label>Figure 1</label>
        <caption>
          <p>Insight assessment across AI models. (A) Nonobvious deduction, (B) discriminative reasoning, and (C) clinical validity. Bars represent the percentage of 200 questions rated as low, moderate, or high insight for ChatGPT, Claude, and AMBOSS.</p>
        </caption>
        <graphic xlink:href="formative_v10i1e100148_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
      </fig>
      <p>For discriminative reasoning, ChatGPT achieved low, moderate, and high ratings in 10.5% (21/200), 60% (120/200), and 29.5% (59/200) of responses, respectively. Claude demonstrated low, moderate, and high ratings in 10% (20/200), 70% (140/200), and 20% (40/200) of responses, respectively. AMBOSS achieved low, moderate, and high ratings in 61.5% (123/200), 33% (66/200), and 5.5% (11/200) of responses, respectively (<xref rid="figure1" ref-type="fig">Figure 1</xref>B).</p>
      <p>Clinical validity was highest for ChatGPT, with 93% (186/200) of responses rated as high and 7% (14/200) rated as low. Claude achieved high and low ratings in 80% (160/200) and 20% (40/200) of responses, respectively, while AMBOSS achieved high and low ratings in 75% (150/200) and 25% (50/200) of responses, respectively (<xref rid="figure1" ref-type="fig">Figure 1</xref>C).</p>
      <p>Substantial agreement was observed between the 2 reviewers. The weighted κ coefficient was 0.67, 0.73, and 0.80 for nonobvious deduction, discriminative reasoning, and clinical validity, respectively.</p>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>In this study, we evaluated the performance of ChatGPT, Claude, and AMBOSS on 200 EBU-style MCQs using a multidimensional assessment framework that examined not only answer accuracy but also concordance, nonobvious deduction, discriminative reasoning, and clinical validity. Although previous studies have demonstrated that LLMs can achieve pass-level or near–pass-level performance on medical and specialty examinations, these evaluations have largely focused on answer correctness and examination outcomes [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. In contrast, our study assessed the quality and transparency of the reasoning underpinning AI-generated responses, with a particular emphasis on adherence to EAU guideline–based practice.</p>
        <p>Previous work supports the growing role of AI in medical education, with GPT-based models demonstrating strong performance across licensing and board examinations [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref23">23</xref>].</p>
        <p>However, most studies have focused primarily on examination accuracy rather than the reasoning processes underlying responses. For example, Vaishya et al [<xref ref-type="bibr" rid="ref24">24</xref>] evaluated AI performance on orthopedic postgraduate examination questions without systematically assessing explanation quality. Similarly, studies of medical question-answering systems and patient-facing chatbots have shown encouraging performance but have also highlighted that answer correctness alone may not adequately capture response quality, transparency, safety, or clinical applicability [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>].</p>
        <p>Assessing guideline adherence is particularly important given the known limitations of AI-generated clinical recommendations. Talyshinskii et al [<xref ref-type="bibr" rid="ref27">27</xref>] found that although GPT-4.0 demonstrated partially accurate diagnostic knowledge in urolithiasis, its surgical planning recommendations were not consistently aligned with EAU guidelines. Additionally, Eldaneen et al [<xref ref-type="bibr" rid="ref28">28</xref>] reported that current LLMs provide readily accessible guidance on LUTS; however, their unsupervised use in clinical decision-making remains a concern and may be considered premature.</p>
        <p>Broader evaluations of generative AI in health care have similarly highlighted concerns regarding reliability, hallucinations, safety, and the need for human oversight [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. Furthermore, Topol [<xref ref-type="bibr" rid="ref31">31</xref>] argued that successful integration of AI into health care depends on maintaining human clinical oversight, while systematic reviews have identified ongoing concerns regarding reliability, interpretability, and real-world applicability [<xref ref-type="bibr" rid="ref32">32</xref>]. Together, these findings emphasize the importance of evaluating not only whether an answer is correct but also whether the underlying reasoning is transparent and clinically appropriate.</p>
        <p>Although all 3 models achieved an accuracy exceeding 70%, important differences emerged when reasoning quality was assessed directly. ChatGPT demonstrated the strongest overall performance, with the highest accuracy, concordance, insight scores, and clinical validity. This finding is consistent with recent reports showing strong performance of GPT-4.0–based models across medical knowledge and clinical reasoning assessments [<xref ref-type="bibr" rid="ref33">33</xref>]. Although all models achieved scores exceeding 70% on the EBU In-Service Assessment workbook, these results should not be interpreted as equivalent to performance on the official EBU part 1 examination because the workbook is an educational resource rather than a validated examination used for certification.</p>
        <p>ChatGPT more frequently incorporated structured, multistep reasoning; guideline-based interpretation; and explicit exclusion of alternative answers, resulting in the highest concordance scores.</p>
        <p>Claude achieved comparable accuracy and demonstrated strong clinical validity and discriminative reasoning. Its explanations frequently justified why alternative answers were incorrect and were generally consistent with accepted urological practice. However, compared with ChatGPT, Claude was less likely to generate reasoning that extended beyond the information explicitly provided in the question stem.</p>
        <p>AMBOSS also demonstrated good overall accuracy and guideline alignment, but its explanations were typically shorter and more descriptive. Although correct answers were often provided, the explanations were less likely to justify the selected option or actively exclude competing alternatives. This was reflected in lower concordance and insight scores than those of ChatGPT and Claude.</p>
        <p>High levels of nonobvious deduction were uncommon across all models. This likely reflects the structure of EBU-style questions, many of which assess factual recall and therefore provide limited opportunities for advanced reasoning. Higher ratings were generally observed in questions requiring the integration of multiple clinical concepts or the application of guideline-based management decisions.</p>
        <p>The greatest differences between models were observed in discriminative reasoning. ChatGPT and Claude more frequently compared competing answer options and explicitly explained why incorrect alternatives should be excluded. In contrast, AMBOSS often identified the correct answer without fully articulating the reasoning process that led to that conclusion. The ability to actively exclude alternative options represents an important component of both examination performance and clinical decision-making and may enhance the educational value of AI-generated explanations.</p>
        <p>All 3 models demonstrated strong clinical validity overall, with most responses aligning with accepted urological practice and EAU guideline recommendations. ChatGPT demonstrated the highest consistency in this domain. These findings are reassuring from an educational perspective, suggesting that clinically inappropriate or potentially misleading recommendations were relatively uncommon across all 3 platforms.</p>
        <p>Our findings provide further insight into previously reported limitations of AI performance in medical assessments. Although studies such as those by Sadeq et al [<xref ref-type="bibr" rid="ref34">34</xref>] have demonstrated variability in chatbot performance across question types and levels of complexity, our results suggest that differences between models are not fully explained by accuracy alone. Models with similar examination performance may differ substantially in how they justify their answers, exclude alternative options, and maintain alignment with guideline-based practice. This observation is consistent with the work of Holzinger et al [<xref ref-type="bibr" rid="ref35">35</xref>], who emphasized the importance of evaluating explanations rather than outputs alone.</p>
        <p>On the basis of the findings of this study, AMBOSS may be particularly useful for factual revision, whereas ChatGPT and Claude may offer additional value for examination preparation because of their more detailed reasoning processes. Nevertheless, none of the evaluated models should be considered substitutes for clinical supervision or contemporary guideline consultation.</p>
        <p>Overall, AI should be introduced in a staged and supervised manner, supporting rather than replacing traditional educational approaches, consistent with recent recommendations regarding the integration of LLMs into medical education [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>].</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>Several limitations should be acknowledged. This study was limited to single-best-answer MCQs and did not assess performance in more complex or multimodal scenarios, such as imaging interpretation, operative planning, or real-world clinical decision-making. The EBU In-Service Assessment workbook was used as an educational benchmark and is not equivalent to the official EBU part 1 examination; therefore, the findings should not be interpreted as reflecting performance on the certification examination itself. Although predefined criteria, assessor calibration, and dual independent raters were used, some subjectivity is unavoidable. In addition, AI models are dynamic systems whose outputs may change over time because of model updates, and responses were evaluated at a single time point without assessment of repeated-run variability or reproducibility. Potential overlap between model training data and educational resources used in this study cannot be excluded. Furthermore, AI models remain dependent on the scope and currency of their training data, raising concerns regarding alignment with evolving clinical guidelines. Finally, performance in a controlled examination setting may not reflect real-world clinical use.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>AI tools demonstrate strong performance on postgraduate urology examination-style questions but differ in reasoning quality and educational value. ChatGPT showed the highest combined accuracy, concordance, and insight, with consistent guideline adherence and strong discriminative reasoning. Claude demonstrated comparable accuracy with good clinical validity but less consistent deductive depth. AMBOSS provided accurate, guideline-aligned responses but relied more on descriptive knowledge with limited explicit reasoning. These findings support the structured and critical use of AI as an adjunct in urology education, emphasizing tools that promote transparent, guideline-based clinical reasoning.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Compliance with the METRICS (Modern Quality Assessment Framework for Evaluating Generative AI Studies in Healthcare) checklist.</p>
        <media xlink:href="formative_v10i1e100148_app1.docx" xlink:title="DOCX File , 17 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Nonobvious deduction, discriminative reasoning, and clinical validity insight levels across AI models.</p>
        <media xlink:href="formative_v10i1e100148_app2.docx" xlink:title="DOCX File , 15 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">EAU</term>
          <def>
            <p>European Association of Urology</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">EBU</term>
          <def>
            <p>European Board of Urology</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">FRCS</term>
          <def>
            <p>Fellowship of the Royal College of Surgeons</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">MCQ</term>
          <def>
            <p>multiple-choice question</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">METRICS</term>
          <def>
            <p>Modern Quality Assessment Framework for Evaluating Generative AI Studies in Healthcare</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">USMLE</term>
          <def>
            <p>United States Medical Licensing Examination</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors declare the use of generative AI (GAI) in the research and writing process. According to the Generative Artificial Intelligence Delegation Taxonomy (2025), the following tasks were delegated to GAI tools under full human supervision: text generation, proofreading and editing, and summarizing text. The GAI tool used was ChatGPT (version 5.5; OpenAI). Responsibility for the final manuscript lies entirely with the authors. GAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>All data generated or analyzed during this study are included in this published article.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>The authors declare that no financial support was received for this study.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Data analysis: ME, SH, OR</p>
        <p>Data collection: ME, SH</p>
        <p>Protocol and project development: KHP</p>
        <p>Writing—original draft: ME, SH, KHP</p>
        <p>Writing—review and editing: ME, SH, OR, PN, AM, HMA, KHP</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="web">
          <article-title>Fellow of the European Board of Urology</article-title>
          <source>European Board of Urology</source>
          <access-date>2026-07-07</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.ebu.com/examination/febu/">https://www.ebu.com/examination/febu/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Boscardin</surname>
              <given-names>CK</given-names>
            </name>
            <name name-style="western">
              <surname>Gin</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Golde</surname>
              <given-names>PB</given-names>
            </name>
            <name name-style="western">
              <surname>Hauer</surname>
              <given-names>KE</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT and generative artificial intelligence for medical education: potential impact and opportunity</article-title>
          <source>Acad Med</source>
          <year>2024</year>
          <month>01</month>
          <day>01</day>
          <volume>99</volume>
          <issue>1</issue>
          <fpage>22</fpage>
          <lpage>7</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/academicmedicine/article-lookup/doi/10.1097/ACM.0000000000005439"/>
          </comment>
          <pub-id pub-id-type="doi">10.1097/ACM.0000000000005439</pub-id>
          <pub-id pub-id-type="medline">37651677</pub-id>
          <pub-id pub-id-type="pii">00001888-202401000-00011</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Qin</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>AI in medical education: medical student perception, curriculum recommendations and design suggestions</article-title>
          <source>BMC Med Educ</source>
          <year>2023</year>
          <month>11</month>
          <day>09</day>
          <volume>23</volume>
          <issue>1</issue>
          <fpage>852</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-023-04700-8"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-023-04700-8</pub-id>
          <pub-id pub-id-type="medline">37946176</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-023-04700-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC10637014</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Alkaissi</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>McFarlane</surname>
              <given-names>SI</given-names>
            </name>
          </person-group>
          <article-title>Artificial hallucinations in ChatGPT: implications in scientific writing</article-title>
          <source>Cureus</source>
          <year>2023</year>
          <month>02</month>
          <day>19</day>
          <volume>15</volume>
          <issue>2</issue>
          <fpage>e35179</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/36811129"/>
          </comment>
          <pub-id pub-id-type="doi">10.7759/cureus.35179</pub-id>
          <pub-id pub-id-type="medline">36811129</pub-id>
          <pub-id pub-id-type="pmcid">PMC9939079</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kung</surname>
              <given-names>TH</given-names>
            </name>
            <name name-style="western">
              <surname>Cheatham</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Medenilla</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sillos</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>De Leon</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Elepaño</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Madriaga</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Aggabao</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Diaz-Candido</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Maningo</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tseng</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title>
          <source>PLOS Digit Health</source>
          <year>2023</year>
          <month>02</month>
          <day>9</day>
          <volume>2</volume>
          <issue>2</issue>
          <fpage>e0000198</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pdig.0000198"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id>
          <pub-id pub-id-type="medline">36812645</pub-id>
          <pub-id pub-id-type="pii">PDIG-D-22-00371</pub-id>
          <pub-id pub-id-type="pmcid">PMC9931230</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gilson</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Safranek</surname>
              <given-names>CW</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Socrates</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Chi</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Taylor</surname>
              <given-names>RA</given-names>
            </name>
            <name name-style="western">
              <surname>Chartash</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>How does ChatGPT perform on the United States Medical Licensing Examination (USMLE)? The implications of large language models for medical education and knowledge assessment</article-title>
          <source>JMIR Med Educ</source>
          <year>2023</year>
          <month>02</month>
          <day>08</day>
          <volume>9</volume>
          <fpage>e45312</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mededu.jmir.org/2023//e45312/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/45312</pub-id>
          <pub-id pub-id-type="medline">36753318</pub-id>
          <pub-id pub-id-type="pii">v9i1e45312</pub-id>
          <pub-id pub-id-type="pmcid">PMC9947764</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Katz</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Bommarito</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Arredondo</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>GPT-4 passes the bar exam</article-title>
          <source>Philos Trans A Math Phys Eng Sci</source>
          <year>2024</year>
          <month>04</month>
          <day>15</day>
          <volume>382</volume>
          <issue>2270</issue>
          <fpage>20230254</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38403056"/>
          </comment>
          <pub-id pub-id-type="doi">10.1098/rsta.2023.0254</pub-id>
          <pub-id pub-id-type="medline">38403056</pub-id>
          <pub-id pub-id-type="pmcid">PMC10894685</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Touma</surname>
              <given-names>NJ</given-names>
            </name>
            <name name-style="western">
              <surname>Caterini</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liblk</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Is ChatGPT ready for primetime? Performance of artificial intelligence on a simulated Canadian urology board exam</article-title>
          <source>Can Urol Assoc J</source>
          <year>2024</year>
          <month>10</month>
          <volume>18</volume>
          <issue>10</issue>
          <fpage>329</fpage>
          <lpage>32</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.5489/cuaj.8800"/>
          </comment>
          <pub-id pub-id-type="doi">10.5489/cuaj.8800</pub-id>
          <pub-id pub-id-type="medline">38896484</pub-id>
          <pub-id pub-id-type="pii">cuaj.8800</pub-id>
          <pub-id pub-id-type="pmcid">PMC11477513</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hetz</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Carl</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Haggenmüller</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Wies</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Kather</surname>
              <given-names>JN</given-names>
            </name>
            <name name-style="western">
              <surname>Michel</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Wessels</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Brinker</surname>
              <given-names>TJ</given-names>
            </name>
          </person-group>
          <article-title>Superhuman performance on urology board questions using an explainable language model enhanced with European Association of Urology guidelines</article-title>
          <source>ESMO Real World Data Digit Oncol</source>
          <year>2024</year>
          <month>10</month>
          <day>04</day>
          <volume>6</volume>
          <fpage>100078</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2949-8201(24)00056-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.esmorw.2024.100078</pub-id>
          <pub-id pub-id-type="medline">41646097</pub-id>
          <pub-id pub-id-type="pii">S2949-8201(24)00056-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC12836625</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bientzle</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hircin</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Kimmerle</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Knipfer</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Smeets</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Gaudin</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Holtz</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Association of online learning behavior and learning outcomes for medical students: large-scale usage data analysis</article-title>
          <source>JMIR Med Educ</source>
          <year>2019</year>
          <month>08</month>
          <day>21</day>
          <volume>5</volume>
          <issue>2</issue>
          <fpage>e13529</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mededu.jmir.org/2019/2/e13529/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/13529</pub-id>
          <pub-id pub-id-type="medline">31436166</pub-id>
          <pub-id pub-id-type="pii">v5i2e13529</pub-id>
          <pub-id pub-id-type="pmcid">PMC6724501</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="web">
          <article-title>Guidelines</article-title>
          <source>European Association of Urology</source>
          <access-date>2026-07-07</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://uroweb.org/guidelines">https://uroweb.org/guidelines</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sallam</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Barakat</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sallam</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>A preliminary checklist (METRICS) to standardize the design and reporting of studies on generative artificial intelligence-based models in health care education and practice: development study involving a literature review</article-title>
          <source>Interact J Med Res</source>
          <year>2024</year>
          <month>02</month>
          <day>15</day>
          <volume>13</volume>
          <fpage>e54704</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.i-jmr.org/2024//e54704/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/54704</pub-id>
          <pub-id pub-id-type="medline">38276872</pub-id>
          <pub-id pub-id-type="pii">v13i1e54704</pub-id>
          <pub-id pub-id-type="pmcid">PMC10905357</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="web">
          <source>ChatGPT</source>
          <access-date>2026-07-20</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://chatgpt.com/">https://chatgpt.com/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="web">
          <source>Claude</source>
          <access-date>2026-07-20</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://claude.ai/login">https://claude.ai/login</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="web">
          <source>AMBOSS</source>
          <access-date>2026-07-20</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.amboss.com/int">https://www.amboss.com/int</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Downing</surname>
              <given-names>SM</given-names>
            </name>
          </person-group>
          <article-title>Validity: on meaningful interpretation of assessment data</article-title>
          <source>Med Educ</source>
          <year>2003</year>
          <month>09</month>
          <volume>37</volume>
          <issue>9</issue>
          <fpage>830</fpage>
          <lpage>7</lpage>
          <pub-id pub-id-type="doi">10.1046/j.1365-2923.2003.01594.x</pub-id>
          <pub-id pub-id-type="medline">14506816</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Miller</surname>
              <given-names>GE</given-names>
            </name>
          </person-group>
          <article-title>The assessment of clinical skills/competence/performance</article-title>
          <source>Acad Med</source>
          <year>1990</year>
          <month>09</month>
          <volume>65</volume>
          <issue>9 Suppl</issue>
          <fpage>S63</fpage>
          <lpage>7</lpage>
          <pub-id pub-id-type="doi">10.1097/00001888-199009000-00045</pub-id>
          <pub-id pub-id-type="medline">2400509</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cook</surname>
              <given-names>DA</given-names>
            </name>
            <name name-style="western">
              <surname>Beckman</surname>
              <given-names>TJ</given-names>
            </name>
          </person-group>
          <article-title>Current concepts in validity and reliability for psychometric instruments: theory and application</article-title>
          <source>Am J Med</source>
          <year>2006</year>
          <month>02</month>
          <volume>119</volume>
          <issue>2</issue>
          <fpage>166.e7</fpage>
          <lpage>.e16</lpage>
          <pub-id pub-id-type="doi">10.1016/j.amjmed.2005.10.036</pub-id>
          <pub-id pub-id-type="medline">16443422</pub-id>
          <pub-id pub-id-type="pii">S0002-9343(05)01037-5</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chmura Kraemer</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Periyakoil</surname>
              <given-names>VS</given-names>
            </name>
            <name name-style="western">
              <surname>Noda</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Kappa coefficients in medical research</article-title>
          <source>Stat Med</source>
          <year>2002</year>
          <month>07</month>
          <day>30</day>
          <volume>21</volume>
          <issue>14</issue>
          <fpage>2109</fpage>
          <lpage>29</lpage>
          <pub-id pub-id-type="doi">10.1002/sim.1180</pub-id>
          <pub-id pub-id-type="medline">12111890</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of large language models in medical examinations: a scoping review protocol</article-title>
          <source>PLoS One</source>
          <year>2026</year>
          <month>04</month>
          <day>22</day>
          <volume>21</volume>
          <issue>4</issue>
          <fpage>e0347539</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pone.0347539"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0347539</pub-id>
          <pub-id pub-id-type="medline">42018550</pub-id>
          <pub-id pub-id-type="pii">PONE-D-25-35513</pub-id>
          <pub-id pub-id-type="pmcid">PMC13102214</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kollitsch</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Eredics</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Marszalek</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rauchenwald</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Brookman-May</surname>
              <given-names>SD</given-names>
            </name>
            <name name-style="western">
              <surname>Burger</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Körner-Riffard</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>May</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>How does artificial intelligence master urological board examinations? A comparative analysis of different large language models' accuracy and reliability in the 2022 in-service assessment of the European Board of Urology</article-title>
          <source>World J Urol</source>
          <year>2024</year>
          <month>01</month>
          <day>10</day>
          <volume>42</volume>
          <issue>1</issue>
          <fpage>20</fpage>
          <pub-id pub-id-type="doi">10.1007/s00345-023-04749-6</pub-id>
          <pub-id pub-id-type="medline">38197996</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00345-023-04749-6</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Şahin</surname>
              <given-names>MF</given-names>
            </name>
            <name name-style="western">
              <surname>Doğan</surname>
              <given-names>Ç</given-names>
            </name>
            <name name-style="western">
              <surname>Topkaç</surname>
              <given-names>EC</given-names>
            </name>
            <name name-style="western">
              <surname>Şeramet</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tuncer</surname>
              <given-names>FB</given-names>
            </name>
            <name name-style="western">
              <surname>Yazıcı</surname>
              <given-names>CM</given-names>
            </name>
          </person-group>
          <article-title>Which current chatbot is more competent in urological theoretical knowledge? A comparative analysis by the European Board of Urology in-service assessment</article-title>
          <source>World J Urol</source>
          <year>2025</year>
          <month>02</month>
          <day>11</day>
          <volume>43</volume>
          <issue>1</issue>
          <fpage>116</fpage>
          <pub-id pub-id-type="doi">10.1007/s00345-025-05499-3</pub-id>
          <pub-id pub-id-type="medline">39932577</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00345-025-05499-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC11813998</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shieh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Kumar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Freed</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Majety</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Assessing ChatGPT 4.0's test performance and clinical diagnostic accuracy on USMLE STEP 2 CK and clinical case reports</article-title>
          <source>Sci Rep</source>
          <year>2024</year>
          <month>04</month>
          <day>23</day>
          <volume>14</volume>
          <issue>1</issue>
          <fpage>9330</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-024-58760-x"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-024-58760-x</pub-id>
          <pub-id pub-id-type="medline">38654011</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-024-58760-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC11039662</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vaishya</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Iyengar</surname>
              <given-names>KP</given-names>
            </name>
            <name name-style="western">
              <surname>Patralekh</surname>
              <given-names>MK</given-names>
            </name>
            <name name-style="western">
              <surname>Botchu</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Shirodkar</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Jain</surname>
              <given-names>VK</given-names>
            </name>
            <name name-style="western">
              <surname>Vaish</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Scarlat</surname>
              <given-names>MM</given-names>
            </name>
          </person-group>
          <article-title>Effectiveness of AI-powered chatbots in responding to orthopaedic postgraduate exam questions-an observational study</article-title>
          <source>Int Orthop</source>
          <year>2024</year>
          <month>08</month>
          <volume>48</volume>
          <issue>8</issue>
          <fpage>1963</fpage>
          <lpage>9</lpage>
          <pub-id pub-id-type="doi">10.1007/s00264-024-06182-9</pub-id>
          <pub-id pub-id-type="medline">38619565</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00264-024-06182-9</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sayres</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wulczyn</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Amin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hou</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>SR</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Neal</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Rashid</surname>
              <given-names>QM</given-names>
            </name>
            <name name-style="western">
              <surname>Schaekermann</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dash</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NH</given-names>
            </name>
            <name name-style="western">
              <surname>Lachgar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Prakash</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Green</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Dominowska</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Tomašev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>JK</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>DR</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Toward expert-level medical question answering with large language models</article-title>
          <source>Nat Med</source>
          <year>2025</year>
          <month>03</month>
          <volume>31</volume>
          <issue>3</issue>
          <fpage>943</fpage>
          <lpage>50</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id>
          <pub-id pub-id-type="medline">39779926</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-03423-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11922739</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ayers</surname>
              <given-names>JW</given-names>
            </name>
            <name name-style="western">
              <surname>Poliak</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dredze</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Leas</surname>
              <given-names>EC</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Kelley</surname>
              <given-names>JB</given-names>
            </name>
            <name name-style="western">
              <surname>Faix</surname>
              <given-names>DJ</given-names>
            </name>
            <name name-style="western">
              <surname>Goodman</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Longhurst</surname>
              <given-names>CA</given-names>
            </name>
            <name name-style="western">
              <surname>Hogarth</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Smith</surname>
              <given-names>DM</given-names>
            </name>
          </person-group>
          <article-title>Comparing physician and artificial intelligence chatbot responses to patient questions posted to a public social media forum</article-title>
          <source>JAMA Intern Med</source>
          <year>2023</year>
          <month>06</month>
          <day>01</day>
          <volume>183</volume>
          <issue>6</issue>
          <fpage>589</fpage>
          <lpage>96</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37115527"/>
          </comment>
          <pub-id pub-id-type="doi">10.1001/jamainternmed.2023.1838</pub-id>
          <pub-id pub-id-type="medline">37115527</pub-id>
          <pub-id pub-id-type="pii">2804309</pub-id>
          <pub-id pub-id-type="pmcid">PMC10148230</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Talyshinskii</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Juliebø-Jones</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Zeeshan Hameed</surname>
              <given-names>BM</given-names>
            </name>
            <name name-style="western">
              <surname>Naik</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Adhikari</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Zhanbyrbekuly</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Tzelves</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Somani</surname>
              <given-names>BK</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT as a clinical decision maker for urolithiasis: compliance with the current European Association of Urology guidelines</article-title>
          <source>Eur Urol Open Sci</source>
          <year>2024</year>
          <month>09</month>
          <day>16</day>
          <volume>69</volume>
          <fpage>51</fpage>
          <lpage>62</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2666-1683(24)00653-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.euros.2024.08.015</pub-id>
          <pub-id pub-id-type="medline">39318971</pub-id>
          <pub-id pub-id-type="pii">S2666-1683(24)00653-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC11421362</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Eldaneen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Eissa</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sabaa</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Nikolinakos</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Saber-Khalaf</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pang</surname>
              <given-names>KH</given-names>
            </name>
          </person-group>
          <article-title>Accuracy of large language models in answering urological questions on lower urinary tract symptoms: comparison with the EAU 2025 guidelines</article-title>
          <source>Cent European J Urol</source>
          <year>2026</year>
          <volume>79</volume>
          <issue>2</issue>
          <fpage>152</fpage>
          <lpage>60</lpage>
          <pub-id pub-id-type="doi">10.5173/ceju.2026.0011</pub-id>
          <pub-id pub-id-type="medline">42375715</pub-id>
          <pub-id pub-id-type="pii">CEJU-79-02-0011</pub-id>
          <pub-id pub-id-type="pmcid">PMC13312229</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Bubeck</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Petro</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine</article-title>
          <source>N Engl J Med</source>
          <year>2023</year>
          <month>03</month>
          <day>30</day>
          <volume>388</volume>
          <issue>13</issue>
          <fpage>1233</fpage>
          <lpage>9</lpage>
          <pub-id pub-id-type="doi">10.1056/NEJMsr2214184</pub-id>
          <pub-id pub-id-type="medline">36988602</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dave</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Athaluri</surname>
              <given-names>SA</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT in medicine: an overview of its applications, advantages, limitations, future prospects, and ethical considerations</article-title>
          <source>Front Artif Intell</source>
          <year>2023</year>
          <month>5</month>
          <day>4</day>
          <volume>6</volume>
          <fpage>1169595</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37215063"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/frai.2023.1169595</pub-id>
          <pub-id pub-id-type="medline">37215063</pub-id>
          <pub-id pub-id-type="pmcid">PMC10192861</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Topol</surname>
              <given-names>EJ</given-names>
            </name>
          </person-group>
          <article-title>High-performance medicine: the convergence of human and artificial intelligence</article-title>
          <source>Nat Med</source>
          <year>2019</year>
          <month>01</month>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>44</fpage>
          <lpage>56</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-018-0300-7</pub-id>
          <pub-id pub-id-type="medline">30617339</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-018-0300-7</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sallam</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT utility in healthcare education, research, and practice: systematic review on the promising perspectives and valid concerns</article-title>
          <source>Healthcare (Basel)</source>
          <year>2023</year>
          <month>03</month>
          <day>19</day>
          <volume>11</volume>
          <issue>6</issue>
          <fpage>887</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=healthcare11060887"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/healthcare11060887</pub-id>
          <pub-id pub-id-type="medline">36981544</pub-id>
          <pub-id pub-id-type="pii">healthcare11060887</pub-id>
          <pub-id pub-id-type="pmcid">PMC10048148</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bicknell</surname>
              <given-names>BT</given-names>
            </name>
            <name name-style="western">
              <surname>Butler</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Whalen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ricks</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Dixon</surname>
              <given-names>CJ</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Spaedy</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Skelton</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Edupuganti</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Dzubinski</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Tate</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Dyess</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Lindeman</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Lehmann</surname>
              <given-names>LS</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT-4 Omni performance in USMLE disciplines and clinical skills: comparative analysis</article-title>
          <source>JMIR Med Educ</source>
          <year>2024</year>
          <month>11</month>
          <day>06</day>
          <volume>10</volume>
          <fpage>e63430</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mededu.jmir.org/2024//e63430/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/63430</pub-id>
          <pub-id pub-id-type="medline">39504445</pub-id>
          <pub-id pub-id-type="pii">v10i1e63430</pub-id>
          <pub-id pub-id-type="pmcid">PMC11611793</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sadeq</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Ghorab</surname>
              <given-names>RM</given-names>
            </name>
            <name name-style="western">
              <surname>Ashry</surname>
              <given-names>MH</given-names>
            </name>
            <name name-style="western">
              <surname>Abozaid</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Banihani</surname>
              <given-names>HA</given-names>
            </name>
            <name name-style="western">
              <surname>Salem</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Aisheh</surname>
              <given-names>MT</given-names>
            </name>
            <name name-style="western">
              <surname>Abuzahra</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mourid</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Assker</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Ayyad</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Moawad</surname>
              <given-names>MH</given-names>
            </name>
          </person-group>
          <article-title>AI chatbots show promise but limitations on UK medical exam questions: a comparative performance study</article-title>
          <source>Sci Rep</source>
          <year>2024</year>
          <month>08</month>
          <day>14</day>
          <volume>14</volume>
          <issue>1</issue>
          <fpage>18859</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-024-68996-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-024-68996-2</pub-id>
          <pub-id pub-id-type="medline">39143077</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-024-68996-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC11324724</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Holzinger</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Carrington</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Müller</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Measuring the quality of explanations: the System Causability Scale (SCS): comparing human and machine explanations</article-title>
          <source>Kunstliche Intell (Oldenbourg)</source>
          <year>2020</year>
          <volume>34</volume>
          <issue>2</issue>
          <fpage>193</fpage>
          <lpage>8</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/32549653"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s13218-020-00636-z</pub-id>
          <pub-id pub-id-type="medline">32549653</pub-id>
          <pub-id pub-id-type="pii">636</pub-id>
          <pub-id pub-id-type="pmcid">PMC7271052</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Abd-Alrazaq</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>AlSaad</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Alhuwail</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Ahmed</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Healy</surname>
              <given-names>PM</given-names>
            </name>
            <name name-style="western">
              <surname>Latifi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Aziz</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Damseh</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Alabed Alrazak</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sheikh</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Large language models in medical education: opportunities, challenges, and future directions</article-title>
          <source>JMIR Med Educ</source>
          <year>2023</year>
          <month>06</month>
          <day>01</day>
          <volume>9</volume>
          <fpage>e48291</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mededu.jmir.org/2023//e48291/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/48291</pub-id>
          <pub-id pub-id-type="medline">37261894</pub-id>
          <pub-id pub-id-type="pii">v9i1e48291</pub-id>
          <pub-id pub-id-type="pmcid">PMC10273039</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Masters</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence in medical education</article-title>
          <source>Med Teach</source>
          <year>2019</year>
          <month>09</month>
          <volume>41</volume>
          <issue>9</issue>
          <fpage>976</fpage>
          <lpage>80</lpage>
          <pub-id pub-id-type="doi">10.1080/0142159X.2019.1595557</pub-id>
          <pub-id pub-id-type="medline">31007106</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
