<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JFR</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id>
      <journal-title>JMIR Formative Research</journal-title>
      <issn pub-type="epub">2561-326X</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v10i1e95883</article-id>
      <article-id pub-id-type="pmid">42647860</article-id>
      <article-id pub-id-type="doi">10.2196/95883</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Large Language Models for Patient Education in Cardiovascular Imaging: Prospective Observational Comparative Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>MacNeill</surname>
            <given-names>Luke</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Mondal</surname>
            <given-names>Anish</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Marey</surname>
            <given-names>Ahmed</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-3659-1696</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Pal</surname>
            <given-names>Basudha</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff02" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0009-0920-8565</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Yaşar</surname>
            <given-names>Ayşenur Buz</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff03" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1324-2810</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Rath</surname>
            <given-names>Shree</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff04" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-4273-0827</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Francese</surname>
            <given-names>Giulia</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff05" ref-type="aff">5</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0008-0785-5542</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>M Ghorab</surname>
            <given-names>Hossam</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff06" ref-type="aff">6</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-6816-5894</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Niemierko</surname>
            <given-names>Julia</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff07" ref-type="aff">7</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-7124-3355</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author">
          <name name-style="western">
            <surname>Jamal</surname>
            <given-names>Muhammad Shah Wali</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff08" ref-type="aff">8</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-5200-2528</ext-link>
        </contrib>
        <contrib id="contrib9" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Umair</surname>
            <given-names>Muhammad</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff09" ref-type="aff">9</xref>
          <address>
            <institution>Columbia University Irving Medical Center</institution>
            <addr-line>630 W. 168th St.</addr-line>
            <addr-line>New York, NY, 10032</addr-line>
            <country>United States</country>
            <phone>1 212 305 2862</phone>
            <email>mu2331@cumc.columbia.edu</email>
          </address>
          <xref rid="aff10" ref-type="aff">10</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-6113-8335</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff01">
        <label>1</label>
        <institution>Shaikh Khalifa Medical City</institution>
        <addr-line>Abu Dhabi, Abu Dhabi</addr-line>
        <country>United Arab Emirates</country>
      </aff>
      <aff id="aff02">
        <label>2</label>
        <institution>Johns Hopkins University</institution>
        <addr-line>Baltimore, MD</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff03">
        <label>3</label>
        <institution>Bolu Abant İzzet Baysal University</institution>
        <addr-line>Bolu, Bolu</addr-line>
        <country>Turkey</country>
      </aff>
      <aff id="aff04">
        <label>4</label>
        <institution>All India Institute of Medical Sciences Bhubaneswar</institution>
        <addr-line>Bhubaneshwar, Odisha</addr-line>
        <country>India</country>
      </aff>
      <aff id="aff05">
        <label>5</label>
        <institution>Centre Hospitalier Universitaire de Rouen</institution>
        <addr-line>Rouen, Normandy</addr-line>
        <country>France</country>
      </aff>
      <aff id="aff06">
        <label>6</label>
        <institution>Alexandria University</institution>
        <addr-line>Alexandria, Alexandria</addr-line>
        <country>Egypt</country>
      </aff>
      <aff id="aff07">
        <label>7</label>
        <institution>Gdańsk Medical University</institution>
        <addr-line>Gdansk, Pomerania</addr-line>
        <country>Poland</country>
      </aff>
      <aff id="aff08">
        <label>8</label>
        <institution>King Edward Medical University</institution>
        <addr-line>Lahore, Punjab</addr-line>
        <country>Pakistan</country>
      </aff>
      <aff id="aff09">
        <label>9</label>
        <institution>Columbia University Irving Medical Center</institution>
        <addr-line>New York, NY</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff10">
        <label>10</label>
        <institution>Johns Hopkins Medicine</institution>
        <addr-line>Baltimore, MD</addr-line>
        <country>United States</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Muhammad Umair <email>mu2331@cumc.columbia.edu</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>26</day>
        <month>8</month>
        <year>2026</year>
      </pub-date>
      <volume>10</volume>
      <elocation-id>e95883</elocation-id>
      <history>
        <date date-type="received">
          <day>22</day>
          <month>3</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>8</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>18</day>
          <month>7</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>21</day>
          <month>7</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Ahmed Marey, Basudha Pal, Ayşenur Buz Yaşar, Shree Rath, Giulia Francese, Hossam M Ghorab, Julia Niemierko, Muhammad Shah Wali Jamal, Muhammad Umair. Originally published in JMIR Formative Research (https://formative.jmir.org), 26.08.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on https://formative.jmir.org, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://formative.jmir.org/2026/1/e95883" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Large language models (LLMs) are increasingly used to support digital health communication, yet their reliability in patient-facing cardiovascular imaging education remains uncertain. Cardiovascular imaging involves complex terminology and procedural details that many patients struggle to understand, creating a need for accurate, clear, and reassuring explanations. While prior evaluations of conversational AI have focused primarily on diagnostic reasoning or clinician-oriented tasks, few studies have systematically compared contemporary LLMs in their ability to communicate effectively with patients.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to compare the accuracy, clarity, completeness, and patient-centered communication quality of responses generated by 3 state-of-the-art conversational agents (DeepSeek, GPT-o1, and GPT-4o) when addressing real-world patient questions about cardiovascular imaging.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>A prospective methodological evaluation was conducted using 84 unique patient-centered questions curated from authoritative cardiovascular information sources and online patient forums. Each question was independently submitted to DeepSeek, GPT-o1, and GPT-4o in isolated sessions to avoid contextual contamination. Two cardiovascular radiologists scored each response across 4 domains (accuracy, clarity and appropriateness, completeness, and user engagement and reassurance) using a standardized 3-point rubric (total score range 4-12). Discrepancies were resolved through predefined adjudication procedures. Because the scores were ordinal, median domain and composite scores with IQRs were summarized and compared across the 3 models using the Kruskal-Wallis test, with ε<sup>2</sup> as an effect size. Statistical significance was defined as an α value of .05.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Across the 84 patient questions, all 3 models produced largely accurate, clear, and complete responses, with comparably high scores across the accuracy, clarity, and completeness domains (median 3 of 3, IQR 3-3 in each). The only meaningful difference appeared in user engagement and reassurance. A “good” engagement rating was assigned to 96.4% (81/84) of DeepSeek responses and 98.8% (83/84) of GPT-o1 responses but only 53.6% (45/84) of GPT-4o responses (Kruskal-Wallis <italic>P</italic>&#60;.001). Composite scores were correspondingly lower for GPT-4o (median 11, IQR 10-12) than for DeepSeek and GPT-o1 (both median 12, IQR 11-12; <italic>P</italic>&#60;.001). No significant differences were observed across models for accuracy (<italic>P</italic>=.91), clarity (<italic>P</italic>=.06), or completeness (<italic>P</italic>=.65), and no unsafe statements were identified in any model.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>DeepSeek and GPT-o1 consistently delivered accurate, clear, and patient-centered explanations of cardiovascular imaging questions, whereas GPT-4o, despite comparable technical accuracy, provided less engaging and reassuring communication. These findings suggest that affective qualities rather than factual correctness represent the main differentiator among current LLMs in patient education tasks. As conversational agents become integrated into cardiovascular imaging workflows, attention to communication tone, emotional support, and health literacy alignment will be essential to ensure safe and effective patient use.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>large language models</kwd>
        <kwd>LLMs</kwd>
        <kwd>DeepSeek</kwd>
        <kwd>GPT-4o</kwd>
        <kwd>GPT-o1</kwd>
        <kwd>cardiovascular imaging</kwd>
        <kwd>patient education</kwd>
        <kwd>artificial intelligence</kwd>
        <kwd>AI</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>AI-driven conversational systems have emerged as influential tools within digital health, where they are increasingly used to support patient education, supplement clinical communication, and provide accessible explanations of complex medical procedures. Large language models (LLMs) such as the generative pretrained transformer series are built on the transformer architecture introduced by Vaswani et al [<xref ref-type="bibr" rid="ref1">1</xref>], which enables the modeling of long-range contextual relationships in text. Subsequent advances in generative pretraining and few-shot learning have further strengthened the capacity of these systems to produce coherent, contextually aligned, and clinically relevant responses [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>].</p>
      <p>In radiology, LLMs have been explored for tasks including automated report generation, radiologic decision support, examination preparation, and patient-facing explanations of imaging results [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. Studies demonstrate that GPT-4 significantly improves upon GPT-3.5 in accuracy and reasoning when tested on radiology board–style assessments [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. These advances have encouraged interest in using LLMs to facilitate patient education, particularly when time constraints and health communication burdens limit the depth of clinician-patient discussions. Patients frequently seek supplemental information online, and conversational agents are well positioned to address questions about imaging procedures that are otherwise difficult to understand.</p>
      <p>DeepSeek [<xref ref-type="bibr" rid="ref11">11</xref>] represents another class of conversational AI systems built using a mixture-of-experts architecture designed to improve computational efficiency and flexibility [<xref ref-type="bibr" rid="ref12">12</xref>]. Early evaluations suggest potential clinical utility in imaging-related assistance, workflow support, patient guidance, and documentation [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. As open-source models become more capable, they may broaden access to AI-assisted education in regions with limited digital infrastructure or fewer health care specialists.</p>
      <p>Cardiovascular diseases remain the leading cause of morbidity and mortality worldwide, and cardiovascular imaging is integral to diagnosis, risk stratification, and therapeutic planning [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. Modalities such as echocardiography, coronary computed tomography angiography, cardiac magnetic resonance imaging, and nuclear imaging provide essential clinical information but are often difficult for patients to interpret. Low health literacy, limited patient familiarity with imaging terminology, and variable clinician communication styles contribute to gaps in understanding that can hinder informed decision-making. Prior research in digital health highlights the importance of accessible patient education tools, particularly in specialized domains where information needs are high and health literacy challenges are common [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref22">22</xref>].</p>
      <p>Although conversational agents show promise as supplemental educational tools, concerns remain regarding variability in accuracy, potential for hallucinations, inconsistent depth of explanation, and differences in empathetic communication [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. These limitations are especially relevant in cardiovascular imaging, where misinterpretation of procedural details, risks, or preparation steps may lead to confusion or anxiety. Despite increasing public reliance on AI-generated medical information, few studies have systematically evaluated how contemporary LLMs respond to patient-centered questions specifically related to cardiovascular imaging. Prior evaluations of LLMs in radiology and patient education have consistently demonstrated strong performance in factual accuracy, clinical coherence, and alignment with established guidelines [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>]. Contemporary systems such as DeepSeek, GPT-o1, and GPT-4o, therefore, reflect a level of technical maturity that supports their potential use as adjunct patient education tools. However, as accuracy and completeness have become increasingly comparable across state-of-the-art models, emerging differences are more likely to arise in affective and communicative dimensions rather than technical correctness alone. Empathy, reassurance, and tone are central to effective patient education, particularly in high-stakes settings such as cardiovascular imaging, where patient anxiety and uncertainty are common.</p>
      <p>Building on our prior evaluation of earlier-generation language models, which emphasized overall response reliability [<xref ref-type="bibr" rid="ref26">26</xref>], the present study examined a much larger and more diverse set of real-world patient questions and focused on state-of-the-art architectures (GPT-o1 [OpenAI] [<xref ref-type="bibr" rid="ref27">27</xref>], GPT-4o [Open AI] [<xref ref-type="bibr" rid="ref28">28</xref>], and DeepSeek [<xref ref-type="bibr" rid="ref11">11</xref>]). This study also incorporated a refined evaluation rubric with explicit emphasis on user engagement and reassurance and quantitatively assessed affective communication differences using categorical and composite analyses. By centering the evaluation on real patient questions and examining communication quality rather than exam-style knowledge, this study addressed an underexplored aspect of how LLMs function in everyday clinical communication.</p>
      <p>Accordingly, this study aimed to (1) assess the accuracy, clarity, and completeness of responses generated by GPT-o1, GPT-4o, and DeepSeek to patient-oriented cardiovascular imaging questions; (2) evaluate the user engagement and reassurance conveyed in these responses, including their capacity to support understanding and reduce procedure-related anxiety; (3) compare performance across the 3 models using a standardized evaluation framework; and (4) consider their feasibility as adjunct patient education tools, particularly in settings with limited specialist availability or health literacy challenges.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Study Design</title>
        <p>This prospective methodological study evaluated the performance of 3 LLMs in generating patient-oriented information about cardiovascular imaging. We followed the STROBE (Strengthening the Reporting of Observational Studies in Epidemiology) checklist for observational studies from the EQUATOR (Enhancing the Quality and Transparency of Health Research) network [<xref ref-type="bibr" rid="ref29">29</xref>] and have provided it in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>This study did not involve human participants, the collection or analysis of identifiable private patient information, or any clinical intervention. All inputs were publicly available patient education materials, and all outputs were AI-generated text. The study, therefore, did not meet the regulatory definition of human subject research, and prospective institutional review board or research ethics board approval was not sought. This determination is consistent with the US Department of Health and Human Services Common Rule (Title 45 of the Code of Federal Regulations §46.102(e) [<xref ref-type="bibr" rid="ref30">30</xref>], under which an activity constitutes human subject research only when an investigator obtains information or biospecimens through interaction with living individuals or obtains identifiable private information, and with the approach taken in prior evaluations of conversational agents in health care communication [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Because no human participants were enrolled and no identifiable data were used, informed consent was not applicable.</p>
      </sec>
      <sec>
        <title>Question Selection and Data Acquisition</title>
        <p>Patient-centered questions were compiled from established cardiovascular information resources, including the American College of Cardiology, the American Heart Association, and MedlinePlus. These sources provide standardized, guideline-aligned educational material for patients with cardiovascular diseases [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. To ensure broad representation of real-world patient concerns, additional questions were gathered from publicly accessible cardiovascular support forums and online patient communities, reflecting common uncertainties expressed in digital health environments.</p>
        <p>All collected questions underwent independent review by 3 authors with experience in cardiovascular imaging and patient communication. During this review, questions that were repetitive, ambiguous, subjective, or not directly related to cardiovascular imaging were excluded. The final dataset consisted of 84 unique patient-oriented questions. The use of structured, curated question sets follows established approaches in prior LLM evaluation studies examining accuracy and patient communication quality [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Minor edits were applied to improve grammar and readability while preserving original meaning. The included questions encompassed procedural preparation, imaging risks, diagnostic purpose, radiation exposure, and lifestyle considerations associated with cardiovascular imaging.</p>
      </sec>
      <sec>
        <title>Response Generation Procedure</title>
        <p>Each of the 84 questions was submitted to 3 LLMs: GPT-4o, GPT-o1, and DeepSeek. All responses were generated on May 9, 2025. To prevent contextual contamination across queries, each question was entered into a separate, newly initiated chat session, an approach consistent with single-turn evaluation methodologies used in previous radiology-focused LLM studies [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. No follow-up prompts, clarifications, or feedback were provided. All questions were submitted in English. Each response was generated once and was not regenerated. Default platform settings were used for all models, and no generation parameters (eg, temperature or sampling settings) were manually modified by the investigators. <xref rid="figure1" ref-type="fig">Figures 1</xref>-<xref rid="figure3" ref-type="fig">3</xref> show the different responses to one example question from the 3 models. The raw responses to all questions from each model can be found in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendices 2</xref>-<xref ref-type="supplementary-material" rid="app4">4</xref>.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>DeepSeek's response to question 14.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e95883_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>GPT-o1's response to question 14.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e95883_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>GPT-4o's response to question 14.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e95883_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Evaluation Framework</title>
        <p>To evaluate the quality of AI-generated responses to patient inquiries regarding cardiovascular imaging, a structured assessment framework was developed based on established approaches used in prior analyses of conversational agents, medical question–answering systems, and radiology-focused LLM evaluations [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. The framework consisted of 4 primary evaluation criteria: accuracy, clarity and appropriateness, completeness, and user engagement and reassurance.</p>
        <p>Two independent cardiovascular radiologists conducted the evaluations using this standardized scoring rubric. Using multiple expert reviewers and predefined scoring criteria aligns with recommended methodological practices for assessing communication quality and ensuring reproducibility in AI-mediated health information research [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. <xref rid="figure4" ref-type="fig">Figure 4</xref> illustrates the overall methodological workflow of the study, including question selection, response generation by the 3 LLMs, expert evaluation, and statistical analysis.</p>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>Methodological workflow for the comparative evaluation of AI-generated cardiovascular imaging information. This diagram illustrates the 4-stage pipeline designed to assess the clinical utility of large language models in patient education. Q&#38;A: question and answer.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e95883_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Evaluation Criteria and Scoring System</title>
        <sec>
          <title>Overview</title>
          <p>Each AI-generated response was evaluated across the 4 domains (accuracy, clarity and appropriateness, completeness, and user engagement and reassurance) on a scale from 1 to 3, as described in <xref ref-type="table" rid="table1">Table 1</xref>. This multidimensional evaluation approach reflects prior frameworks used to assess patient communication, clinical accuracy, and usability in LLM outputs [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref31">31</xref>].</p>
          <table-wrap position="float" id="table1">
            <label>Table 1</label>
            <caption>
              <p>Scoring rubric for the 4 evaluation domains. Each AI-generated response was rated on a scale from 1 to 3 on every domain.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="30"/>
              <col width="180"/>
              <col width="790"/>
              <thead>
                <tr valign="top">
                  <td colspan="2">Domain and score</td>
                  <td>Description</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td colspan="3">
                    <bold>Accuracy</bold>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3 (excellent)</td>
                  <td>Fully accurate, up-to-date, aligned with established guidelines, and free from medical errors or misleading statements</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>2 (moderate)</td>
                  <td>Mostly accurate, with minor inaccuracies or slightly outdated phrasing that do not substantially affect correctness</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>1 (poor)</td>
                  <td>Contains significant inaccuracies, incorrect medical concepts, or misleading statements that could misinform patients</td>
                </tr>
                <tr valign="top">
                  <td colspan="3">
                    <bold>Clarity and appropriateness</bold>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3 (excellent)</td>
                  <td>Clear and well structured and uses simple language without sacrificing accuracy; free from excessive jargon or overly complex phrasing</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>2 (moderate)</td>
                  <td>Mostly clear but contains some jargon, complex phrasing, or minor ambiguities</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>1 (poor)</td>
                  <td>Poorly structured, contains excessive medical jargon, or an overly simplistic response that lacks necessary details</td>
                </tr>
                <tr valign="top">
                  <td colspan="3">
                    <bold>Completeness</bold>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3 (excellent)</td>
                  <td>Completely answers the question, addressing all relevant aspects comprehensively</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>2 (moderate)</td>
                  <td>Partially answers the question, omitting minor details but still providing useful information</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>1 (poor)</td>
                  <td>Incomplete response; omits key details necessary for understanding</td>
                </tr>
                <tr valign="top">
                  <td colspan="3">
                    <bold>User engagement and reassurance</bold>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3 (excellent)</td>
                  <td>Supportive, reassuring, and encourages informed patient engagement with clinicians</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>2 (moderate)</td>
                  <td>Moderately engaging but lacks strong reassurance or appropriate motivational phrasing</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>1 (poor)</td>
                  <td>Neutral, overly robotic, or anxiety-provoking language that may negatively affect patient confidence</td>
                </tr>
              </tbody>
            </table>
          </table-wrap>
        </sec>
        <sec>
          <title>Accuracy</title>
          <p>Accuracy assesses whether the response is factually correct, consistent with contemporary cardiovascular imaging guidelines (such as those from the American Heart Association and American College of Cardiology), and free from medical errors or misleading statements. Accuracy has been a central outcome in prior LLM performance studies in radiology and clinical education [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. Assessment of accuracy considered alignment with evidence-based medical practices, the presence of factual inaccuracies or outdated concepts, the correct use of medical terminology, and the avoidance of speculation or unverified claims.</p>
        </sec>
        <sec>
          <title>Clarity and Appropriateness</title>
          <p>This domain evaluates whether the response is logically structured and clearly written, uses accessible language, and is appropriate for a general patient audience. Clarity and readability are consistent with digital health literacy and patient education evaluation frameworks [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Assessment of clarity and appropriateness considered the use of patient-friendly language while maintaining medical accuracy, the logical structure and clarity of the explanation, and an appropriate level of detail without excessive complexity.</p>
        </sec>
        <sec>
          <title>Completeness</title>
          <p>Completeness assesses whether the response fully addresses the patient’s inquiry without omitting essential clinical or procedural information. Completeness has been a key quality indicator in prior evaluations of medical LLM outputs [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Assessment of completeness considered the inclusion of all essential aspects of the question and the avoidance of partial explanations that may leave patients uncertain.</p>
        </sec>
        <sec>
          <title>User Engagement and Reassurance</title>
          <p>This domain evaluates whether the response provides emotional support, fosters patient confidence, and encourages appropriate follow-up with health care professionals. This dimension is aligned with patient-centered communication evaluations in conversational AI research [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. Assessment of user engagement and reassurance considered the encouragement of patient empowerment and of further discussion with a health care provider and the avoidance of unnecessary alarmism or overly cautious phrasing.</p>
        </sec>
      </sec>
      <sec>
        <title>Discrepancy Resolution Mechanism</title>
        <p>To minimize evaluator subjectivity, a predefined discrepancy resolution procedure was implemented. This approach mirrors adjudication methodologies used in diagnostic performance research and multi-rater AI evaluation studies [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. Disagreements were resolved through a structured process. The 2 cardiovascular radiologists first scored each response independently. Where scores differed by more than 2 points on any category, the evaluators discussed their rationale to reach consensus. For all other disagreements, an independent cardiovascular imaging expert served as the final arbitrator.</p>
      </sec>
      <sec>
        <title>Final Score Calculation</title>
        <p>The composite score was calculated by summing the scores across the 4 evaluation domains (accuracy, clarity and appropriateness, completeness, and user engagement and reassurance), resulting in a total score ranging from 4 to 12. Composite scores were interpreted as follows: a score of 12 indicated excellent performance, reflecting responses that were fully accurate, clear, comprehensive, and patient centered; scores of 10 to 11 indicated good performance with only minor limitations; scores of 7 to 9 indicated adequate performance requiring some improvement; and scores of 4 to 6 indicated poor performance with substantial deficiencies across one or more evaluation domains.</p>
      </sec>
      <sec>
        <title>Statistical Analysis</title>
        <p>For each response and domain, the value entered into the analysis was the single adjudicated score: in cases in which the 2 primary reviewers agreed, that score was used directly, and in cases in which they disagreed, the score assigned by the senior reviewer (R3) was used. The composite total score for each response was then obtained by summing the 4 adjudicated domain scores (each scored from 1-3), yielding a possible range of 4 to 12. Because the domains were measured on a 3-point ordinal scale, scores for each domain and for the composite total were summarized as medians with IQRs rather than as means with SDs. Differences in performance across the 3 LLMs were assessed using the Kruskal-Wallis test, a nonparametric method appropriate for comparing ordinal scores across 3 independent groups. Interrater reliability was assessed using the Cohen κ with quadratic weighting, which accounts for the ordinal structure of the 3-point scoring rubric and penalizes larger discrepancies more heavily than minor disagreements. Agreement was quantified between each primary reviewer (R1 and R2) and the adjudicating senior cardiovascular radiologist (R3). Agreement between R1 and R2 was not evaluated as the primary objective of the reliability analysis was to assess consistency relative to the final adjudicated reference standard rather than raw concordance between initial reviewers. The adjudicating radiologist (R3), as the most experienced reviewer, resolved all discrepancies and generated the final scores used for downstream analysis; therefore, agreement with R3 was considered the most clinically and methodologically relevant measure of scoring reliability. For each domain, and for the composite score, the Kruskal-Wallis <italic>H</italic> statistic, its df, the associated <italic>P</italic> value, and ε<sup>2</sup> as an effect size estimate are reported. The distribution of ordinal ratings across score categories is also presented as counts and percentages for descriptive purposes only to show how responses were spread across the rating categories within each domain; these percentages were not subjected to separate inferential testing. All statistical analyses were performed using R (version 4.5.2; R Foundation for Statistical Computing), with statistical significance defined as an α value of .05.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Overview</title>
        <p>The evaluation sheets completed independently by the 2 primary cardiovascular radiologists, with adjudication by a third radiologist, can be found in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendices 5</xref>-<xref ref-type="supplementary-material" rid="app10">10</xref>. Evaluations for all 3 models by the remaining reviewer are provided as separate worksheets in <xref ref-type="supplementary-material" rid="app11">Multimedia Appendix 11</xref>.</p>
        <p>Interrater reliability analysis (<xref ref-type="table" rid="table2">Table 2</xref>) demonstrated moderate agreement across all models based on commonly used interpretive benchmarks for the Cohen κ (eg, values of 0.41-0.60 indicating moderate agreement and 0.61-0.80 indicating substantial agreement). Notably, GPT-4o exhibited slightly higher quadratic weighted Cohen κ values than DeepSeek and GPT-o1. This finding likely reflects the relative uniformity and neutrality of GPT-4o’s responses, which may be easier to score consistently, rather than superior communication quality. In contrast, the more expressive and patient-centered responses generated by DeepSeek and GPT-o1 may introduce greater subjective variability between raters despite higher overall performance. <xref ref-type="table" rid="table3">Table 3</xref> presents the descriptive statistics for all 4 scoring domains and the total composite score for each model. These differences were modest and should be interpreted cautiously as interrater reliability reflects consistency of scoring rather than intrinsic quality, clinical accuracy, or educational effectiveness of the model responses. Importantly, the observed agreement levels across all models indicate that the scoring rubric could be applied reproducibly by independent reviewers, supporting the robustness of the evaluation framework rather than implying clinically meaningful distinctions between models.</p>
        <p>GPT-o1 demonstrated a consistently strong performance across all domains. Its responses were frequently rated as fully accurate, comprehensive, and clearly structured. Notably, it achieved the highest level of user engagement and reassurance among the 3 models, indicating a communication style that was patient-friendly and supportive. Overall, GPT-o1 produced balanced, high-quality outputs with strengths in clarity and patient-centered phrasing. DeepSeek showed a similarly favorable performance profile, with particularly strong clarity and readability. The model produced well-organized explanations that were easy for lay users to understand. Accuracy and completeness were comparable to those of GPT-o1, and its user engagement scores were also high, although marginally lower than those of GPT-o1. DeepSeek’s overall performance indicates a reliable ability to deliver clear and reassuring patient education content. GPT-4o demonstrated accuracy and completeness comparable to those of the other 2 models but differed meaningfully in its communication tone. While technically sound, its responses were more neutral and less patient oriented, resulting in lower scores in the user engagement and reassurance domain. This reduced patient-centeredness contributed to a lower total score relative to GPT-o1 and DeepSeek despite similar technical quality.</p>
        <p>Overall, the descriptive statistics in <xref ref-type="table" rid="table3">Table 3</xref> show that all 3 models are capable of producing accurate and complete responses. However, GPT-o1 and DeepSeek received higher ratings than GPT-4o in affective and engagement-related qualities, which may be particularly important in patient-facing educational contexts.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Interrater agreement on the evaluations of the different models.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="240"/>
            <col width="290"/>
            <col width="290"/>
            <col width="180"/>
            <thead>
              <tr valign="top">
                <td>AI model</td>
                <td>Cohen κ (R3 vs R1)<sup>a</sup></td>
                <td>Cohen κ (R3 vs R2)<sup>a</sup></td>
                <td><italic>P</italic> value</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>DeepSeek</td>
                <td>0.564</td>
                <td>0.413</td>
                <td>
                  <italic>&#60;.001</italic>
                  <sup>b</sup>
                </td>
              </tr>
              <tr valign="top">
                <td>GPT-o1</td>
                <td>0.519</td>
                <td>0.582</td>
                <td>
                  <italic>&#60;.001</italic>
                  <sup>b</sup>
                </td>
              </tr>
              <tr valign="top">
                <td>GPT-4o</td>
                <td>0.655</td>
                <td>0.638</td>
                <td>
                  <italic>&#60;.001</italic>
                  <sup>b</sup>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup> R1, R2, and R3 means raters 1, 2, and 3, respectively.</p>
            </fn>
            <fn id="table2fn2">
              <p><sup>b</sup>Statistically significant at α=.05.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Descriptive scores and Kruskal-Wallis comparisons of the 3 large language models across the 4 evaluation domains and the composite score<sup>a</sup>.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="210"/>
            <col width="170"/>
            <col width="160"/>
            <col width="160"/>
            <col width="120"/>
            <col width="90"/>
            <col width="90"/>
            <thead>
              <tr valign="top">
                <td>Domain (score range)</td>
                <td>DeepSeek, median (IQR; range)</td>
                <td>GPT-o1, median (IQR; range)</td>
                <td>GPT-4o, median (IQR; range)</td>
                <td><italic>H</italic> statistic (<italic>df</italic>)</td>
                <td><italic>P</italic> value</td>
                <td>ε<sup>2</sup></td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Accuracy (1-3)</td>
                <td>3 (3-3; 2-3)</td>
                <td>3 (3-3; 2-3)</td>
                <td>3 (3-3; 1-3)</td>
                <td>0.20 (2)</td>
                <td>.91<sup>b</sup></td>
                <td>0.001</td>
              </tr>
              <tr valign="top">
                <td>Clarity and appropriateness (1-3)</td>
                <td>3 (3-3; 1-3)</td>
                <td>3 (3-3; 2-3)</td>
                <td>3 (3-3; 1-3)</td>
                <td>5.60 (2)</td>
                <td>.06<sup>b</sup></td>
                <td>0.022</td>
              </tr>
              <tr valign="top">
                <td>Completeness (1-3)</td>
                <td>3 (3-3; 2-3)</td>
                <td>3 (3-3; 2-3)</td>
                <td>3 (3-3; 2-3)</td>
                <td>0.85 (2)</td>
                <td>.65<sup>b</sup></td>
                <td>0.003</td>
              </tr>
              <tr valign="top">
                <td>User engagement and reassurance (1-3)</td>
                <td>3 (3-3; 2-3)</td>
                <td>3 (3-3; 2-3)</td>
                <td>3 (2-3; 2-3)</td>
                <td>76.64 (2)</td>
                <td>
                  <italic>&#60;.001</italic>
                  <sup>
                    <italic>c</italic>
                  </sup>
                </td>
                <td>0.305</td>
              </tr>
              <tr valign="top">
                <td>Total score (4-12)</td>
                <td>12 (11-12; 10-12)</td>
                <td>12 (11-12; 10-12)</td>
                <td>11 (10-12; 8-12)</td>
                <td>17.37 (2)</td>
                <td>
                  <italic>&#60;.001</italic>
                  <sup>
                    <italic>c</italic>
                  </sup>
                </td>
                <td>0.069</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>Between-model differences were assessed using the Kruskal-Wallis test; ε<sup>2</sup> denotes the effect size. Statistical significance was defined as α=.05.</p>
            </fn>
            <fn id="table3fn2">
              <p><sup>b</sup>Not statistically significant at α=.05.</p>
            </fn>
            <fn id="table3fn3">
              <p><sup>c</sup>Statistically significant at α=.05.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Comparative Classification Results</title>
        <p><xref ref-type="table" rid="table4">Table 4</xref> presents the categorical distribution of scores across the 3 LLMs for descriptive purposes. For the accuracy domain, inaccurate responses were rare and limited to GPT-4o, whereas the proportion of fully accurate responses was similarly high across all 3 models.</p>
        <p>For clarity and appropriateness, most responses from all models were rated as clear and concise, and unclear or confusing responses were rare across all 3 models. For completeness, most responses from every model were rated as comprehensive, with the remainder rated as partially comprehensive. The only domain with a meaningful difference was user engagement and reassurance. DeepSeek and GPT-o1 generated predominantly high-engagement responses, whereas GPT-4o produced a substantially lower proportion of high-engagement responses and a correspondingly higher proportion of moderate ones. This communication-related deficit contributed directly to the lower overall ratings of GPT-4o. Differences were also reflected in total score distributions. DeepSeek and GPT-o1 exhibited the highest proportion of perfect scores, whereas GPT-4o achieved a perfect score less often and showed a higher frequency of midrange scores, indicating that its responses, while technically sound, were less consistently reassuring and supportive for patients.</p>
        <p><xref rid="figure5" ref-type="fig">Figure 5</xref> illustrates the distribution of total composite scores across models. While DeepSeek and GPT-o1 showed highly overlapping score distributions and did not differ significantly from one another, both models achieved higher total scores than GPT-4o. The broader spread and downward shift observed for GPT-4o align with its lower engagement and reassurance ratings, indicating greater variability and less consistently patient-centered communication.</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Evaluation of the responses generated by each large language model across 4 domains (accuracy, clarity and appropriateness, completeness, and user engagement and reassurance) using 3-point ordinal scales.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="310"/>
            <col width="0"/>
            <col width="200"/>
            <col width="0"/>
            <col width="230"/>
            <col width="0"/>
            <col width="230"/>
            <thead>
              <tr valign="top">
                <td colspan="3">Term</td>
                <td colspan="5">Model, n (%)</td>
              </tr>
              <tr valign="top">
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">DeepSeek (n=84)</td>
                <td colspan="2">GPT-o1 (n=84)</td>
                <td>GPT-4o (n=84)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="8">
                  <bold>Accuracy (1-3)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Inaccurate</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">2 (2.4)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Partially accurate</td>
                <td colspan="2">15 (17.9)</td>
                <td colspan="2">13 (15.5)</td>
                <td colspan="2">11 (13.1)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Accurate</td>
                <td colspan="2">69 (82.1)</td>
                <td colspan="2">71 (84.5)</td>
                <td colspan="2">71 (84.5)</td>
              </tr>
              <tr valign="top">
                <td colspan="8">
                  <bold>Clarity and appropriateness (1-3)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Unclear or confusing</td>
                <td colspan="2">1 (1.2)</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">1 (1.2)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Somewhat clear</td>
                <td colspan="2">6 (7.1)</td>
                <td colspan="2">15 (17.9)</td>
                <td colspan="2">17 (20.2)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Clear and concise</td>
                <td colspan="2">77 (91.7)</td>
                <td colspan="2">69 (82.1)</td>
                <td colspan="2">66 (78.6)</td>
              </tr>
              <tr valign="top">
                <td colspan="8">
                  <bold>Completeness (1-3)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Incomprehensive</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">0 (0)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Partially comprehensive</td>
                <td colspan="2">18 (21.4)</td>
                <td colspan="2">14 (16.7)</td>
                <td colspan="2">14 (16.7)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Comprehensive</td>
                <td colspan="2">66 (78.6)</td>
                <td colspan="2">70 (83.3)</td>
                <td colspan="2">70 (83.3)</td>
              </tr>
              <tr valign="top">
                <td colspan="8">
                  <bold>User engagement and reassurance (1-3)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Poor</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">0 (0)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Moderate</td>
                <td colspan="2">3 (3.6)</td>
                <td colspan="2">1 (1.2)</td>
                <td colspan="2">39 (46.4)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Good</td>
                <td colspan="2">81 (96.4)</td>
                <td colspan="2">83 (98.8)</td>
                <td colspan="2">45 (53.6)</td>
              </tr>
              <tr valign="top">
                <td colspan="8">
                  <bold>Total score (4-12; no model had a total score&#60;8)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>8</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">1 (1.2)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>9</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">0 (0)</td>
                <td colspan="2">6 (7.1)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>10</td>
                <td colspan="2">8 (9.5)</td>
                <td colspan="2">9 (10.7)</td>
                <td colspan="2">18 (21.4)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>11</td>
                <td colspan="2">28 (33.3)</td>
                <td colspan="2">25 (29.8)</td>
                <td colspan="2">29 (34.5)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>12</td>
                <td colspan="2">48 (57.1)</td>
                <td colspan="2">50 (59.5)</td>
                <td colspan="2">30 (35.7)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <fig id="figure5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Box plots and raw distributions of total composite scores (4-12) across models. DeepSeek and GPT-o1 showed closely overlapping composite score distributions, whereas GPT-4o showed a lower and more variable distribution. The overall difference across the 3 models was significant on the Kruskal-Wallis test (<italic>P</italic>&#60;.001). NS: not significant.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e95883_fig5.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>This study evaluated the quality of patient-facing cardiovascular imaging information generated by GPT-o1, GPT-4o, and DeepSeek across the domains of accuracy, clarity, completeness, and user engagement. Overall, all 3 models demonstrated similarly strong technical performance, with no statistically significant differences in accuracy, clarity, or completeness, indicating that contemporary LLMs are consistently capable of providing factually correct, clear, and comprehensive explanations of cardiovascular imaging for patients. The primary distinction between the models emerged in the user engagement and reassurance domain, where DeepSeek and GPT-o1 consistently generated more supportive, empathetic, and patient-centered responses than GPT-4o. Although the absolute differences in composite scores were modest, they were driven largely by these communication-related characteristics rather than by differences in technical correctness. These findings suggest that state-of-the-art LLMs have largely converged in their ability to deliver accurate educational content and that communication style, reassurance, and patient-centered language may now represent the principal factors differentiating their suitability for patient education. This distinction is particularly relevant in cardiovascular imaging, where patients often seek explanations of unfamiliar procedures while experiencing uncertainty or anxiety, making both the quality of information and the manner in which it is communicated important components of effective patient education.</p>
      </sec>
      <sec>
        <title>Accuracy, Clarity, and Completeness</title>
        <p>The consistently strong performance observed across the accuracy, clarity, and completeness domains aligns with a growing body of literature demonstrating that modern LLMs can effectively generate medically appropriate educational content for patients. Prior evaluations across multiple medical specialties have reported that leading LLMs frequently provide responses that are accurate, comprehensive, and accessible to nonexpert audiences, supporting their potential utility as patient education resources [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Similar observations have been reported in studies evaluating patient-facing questions in radiology, oncology, cardiology, and surgical care, where LLM-generated responses often achieved high ratings for factual correctness and readability while maintaining language that could be understood by lay users [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref33">33</xref>].</p>
        <p>The absence of meaningful differences between models in these technical domains suggests that all 3 systems were capable of conveying essential information regarding cardiovascular imaging procedures. This finding is encouraging because patient education represents one of the most promising and comparatively lower-risk applications of LLM technology within health care. Unlike diagnostic decision-making, patient education primarily involves translating established clinical knowledge into understandable explanations. The ability of multiple independent models to perform similarly in this setting may therefore reflect increasing maturity of LLMs for educational applications.</p>
      </sec>
      <sec>
        <title>User Engagement and Reassurance</title>
        <p>Although technical performance was largely comparable across models, differences emerged in user engagement and reassurance. Responses perceived as more supportive, empathetic, and patient centered generally received higher overall evaluations despite containing information that was often similar in factual content. This observation is consistent with previous research indicating that patients value communication qualities such as empathy, emotional support, reassurance, and conversational tone in addition to factual accuracy [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref33">33</xref>].</p>
        <p>Importantly, effective patient education extends beyond the transmission of correct information. Patients undergoing cardiovascular imaging procedures may experience anxiety related to diagnostic uncertainty, radiation exposure, contrast administration, procedural discomfort, or potential findings. In these contexts, responses that acknowledge concerns and provide reassurance may contribute meaningfully to patient understanding and satisfaction. Prior studies comparing LLM-generated and clinician-generated responses have similarly reported that communication style and perceived empathy often influence overall evaluations of response quality even when factual content is comparable [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref33">33</xref>].</p>
        <p>These findings suggest that evaluation frameworks for patient-facing AI systems should not focus exclusively on factual correctness. Measures of engagement, reassurance, readability, and patient-centered communication may be equally important when assessing the real-world utility of LLM-generated educational content. Future model development efforts may therefore benefit from explicitly optimizing both informational quality and supportive communication.</p>
      </sec>
      <sec>
        <title>Implications for Patient Education</title>
        <p>Taken together, the findings support the potential role of LLMs as adjunct tools for cardiovascular imaging education. Such systems may help patients obtain preliminary information, reinforce explanations provided by health care professionals, and improve access to educational resources outside clinical encounters. However, their greatest value is likely to arise when they complement rather than replace clinician-patient communication.</p>
        <p>The results further suggest that future evaluations of patient-facing LLMs should incorporate broader measures of communication effectiveness, including patient comprehension, trust, anxiety reduction, and perceived usefulness. Although technical accuracy remains essential, educational success ultimately depends on whether information is communicated in a manner that patients find understandable, reassuring, and actionable.</p>
      </sec>
      <sec>
        <title>Hallucinations and Safety Considerations</title>
        <p>Although no hallucinations or clinically concerning inaccuracies were identified in the responses evaluated in this study, hallucinations remain a recognized limitation of LLMs in health care and have been documented even in high-performing systems evaluated across clinical and biomedical domains [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. Consequently, patient-facing deployment should incorporate appropriate safeguards to minimize the risk of incorrect or unsupported information. Proposed mitigation strategies include retrieval-augmented generation approaches that ground responses in curated evidence sources, structured prompting techniques, citation-aware generation, and ongoing human oversight [<xref ref-type="bibr" rid="ref36">36</xref>-<xref ref-type="bibr" rid="ref38">38</xref>]. Continued monitoring, domain-specific refinement, and transparent reporting of safety limitations will remain important as LLMs become increasingly integrated into health care communication workflows [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>].</p>
      </sec>
      <sec>
        <title>Implications for Low- and Middle-Income Countries</title>
        <p>The ability of LLMs to provide accessible educational information may have particular relevance in settings where access to health care professionals and patient education resources is limited. Prior work has highlighted the potential value of AI-based tools in health care systems facing shortages of trained specialists and educational resources [<xref ref-type="bibr" rid="ref41">41</xref>]. However, the present study did not directly evaluate model performance across different languages, cultures, health care systems, or resource-constrained environments. Differences in digital literacy, internet access, and linguistic availability may substantially influence real-world utility. Consequently, further work is needed before conclusions regarding the applicability of these findings to low- and middle-income countries can be drawn.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>This study has several limitations. First, LLM behavior evolves rapidly as models undergo continuous updates and retraining. These findings represent a snapshot of model behavior at a specific time point and should not be interpreted as immutable performance characteristics. Second, the analysis was limited to English-language questions, which may not generalize to multilingual or culturally adapted settings. Third, while this study evaluated patient-oriented communication, it did not assess clinical reasoning depth or medical decision-making accuracy, which are distinct dimensions of LLM performance. Fourth, even with a structured rubric, the scoring depended on expert judgment, which introduces subjectivity and may not capture all subtleties of communication quality. In particular, domains such as user engagement and reassurance represent inherently subjective constructs, and their evaluation reflects a clinical communication perspective rather than direct end user perception. Patient-based scoring or feedback was not incorporated in the present study, and future work should examine how patients themselves perceive engagement, reassurance, and the overall usefulness of LLM-generated explanations. Finally, despite discussing the potential relevance for low- and middle-income country settings, the present study did not directly evaluate model utility, accessibility, or safety in these environments, and thus, conclusions about their broader global applicability should be interpreted with caution.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>These findings carry several broader implications for the use of LLMs in patient education. As the models differed primarily in the affective and patient-centered quality of their communication rather than in clinical correctness, communication tone, empathy, and reassurance should be considered explicit design and evaluation targets for patient-facing AI systems. Realizing the potential of these tools at scale, particularly in settings where access to specialist education is limited, will require safeguards against hallucinations, strategies that promote supportive and comprehensible responses, and adaptation to diverse health literacy levels and languages. Prospective studies involving patients and evaluating real-world outcomes such as comprehension, anxiety reduction, and decision-making will be essential before these technologies can be safely integrated into routine cardiovascular imaging education.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>STROBE checklist.</p>
        <media xlink:href="formative_v10i1e95883_app1.docx" xlink:title="DOCX File , 32 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Raw responses generated by model 1 to all 84 patient-oriented questions.</p>
        <media xlink:href="formative_v10i1e95883_app2.docx" xlink:title="DOCX File , 123 KB"/>
      </supplementary-material>
      <supplementary-material id="app3">
        <label>Multimedia Appendix 3</label>
        <p>Raw responses generated by model 2 to all 84 patient-oriented questions.</p>
        <media xlink:href="formative_v10i1e95883_app3.docx" xlink:title="DOCX File , 185 KB"/>
      </supplementary-material>
      <supplementary-material id="app4">
        <label>Multimedia Appendix 4</label>
        <p>Raw responses generated by model 3 to all 84 patient-oriented questions.</p>
        <media xlink:href="formative_v10i1e95883_app4.docx" xlink:title="DOCX File , 136 KB"/>
      </supplementary-material>
      <supplementary-material id="app5">
        <label>Multimedia Appendix 5</label>
        <p>Evaluation sheet for the DeepSeek responses completed by the first primary reviewer (R1).</p>
        <media xlink:href="formative_v10i1e95883_app5.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 12 KB"/>
      </supplementary-material>
      <supplementary-material id="app6">
        <label>Multimedia Appendix 6</label>
        <p>Evaluation sheet for the GPT-4o responses completed by the first primary reviewer (R1).</p>
        <media xlink:href="formative_v10i1e95883_app6.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 12 KB"/>
      </supplementary-material>
      <supplementary-material id="app7">
        <label>Multimedia Appendix 7</label>
        <p>Evaluation sheet for the GPT-o1 responses completed by the first primary reviewer (R1).</p>
        <media xlink:href="formative_v10i1e95883_app7.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 12 KB"/>
      </supplementary-material>
      <supplementary-material id="app8">
        <label>Multimedia Appendix 8</label>
        <p>Evaluation sheet for the DeepSeek responses completed by the adjudicating reviewer (R3).</p>
        <media xlink:href="formative_v10i1e95883_app8.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 12 KB"/>
      </supplementary-material>
      <supplementary-material id="app9">
        <label>Multimedia Appendix 9</label>
        <p>Evaluation sheet for the GPT-4o responses completed by the adjudicating reviewer (R3).</p>
        <media xlink:href="formative_v10i1e95883_app9.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 12 KB"/>
      </supplementary-material>
      <supplementary-material id="app10">
        <label>Multimedia Appendix 10</label>
        <p>Evaluation sheet for the GPT-o1 responses completed by the adjudicating reviewer (R3).</p>
        <media xlink:href="formative_v10i1e95883_app10.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 12 KB"/>
      </supplementary-material>
      <supplementary-material id="app11">
        <label>Multimedia Appendix 11</label>
        <p>Evaluation sheets for all 3 models completed by the second primary reviewer (R2), provided as separate worksheets.</p>
        <media xlink:href="formative_v10i1e95883_app11.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 26 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">EQUATOR</term>
          <def>
            <p>Enhancing the Quality and Transparency of Health Research</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">STROBE</term>
          <def>
            <p>Strengthening the Reporting of Observational Studies in Epidemiology</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>Generative AI was not used in the writing process of this manuscript.</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>The authors declared no financial support was received for this work.</p>
      </sec>
    </notes>
    <notes>
      <sec>
        <title>Data Availability</title>
        <p>All the detailed responses from the models and the ratings by reviewers are available as supplementary information.</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Conceptualization: AM, BP, MU</p>
        <p>Data curation: AM, ABY, SR, GF, HMG, JN</p>
        <p>Formal analysis: AM, BP, MSWJ</p>
        <p>Investigation: AM, BP, ABY, SR, GF, HMG, JN, MSWJ, MU</p>
        <p>Methodology: AM, BP, MU</p>
        <p>Project administration: MU</p>
        <p>Supervision: MU</p>
        <p>Validation: AM, BP, ABY, SR, GF, HMG, JN</p>
        <p>Visualization: AM, BP</p>
        <p>Writing—original draft: AM, BP, MSWJ</p>
        <p>Writing—review and editing: AM, BP, ABY, SR, GF, HMG, JN, MSWJ, MU</p>
        <p>All authors critically reviewed and revised the manuscript for important intellectual content; approved the final version for publication; and agreed to be accountable for all aspects of the work, including ensuring that questions related to the accuracy or integrity of the work are appropriately investigated and resolved.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vaswani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shazeer</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Parmar</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Uszkoreit</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jones</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Gomez</surname>
              <given-names>AN</given-names>
            </name>
            <name name-style="western">
              <surname>Kaiser</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Polosukhin</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Attention is all you need</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on June 12, 2017</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Radford</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Narasimhan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Salimans</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Sutskever</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Improving language understanding with unsupervised learning</article-title>
          <source>OpenAI</source>
          <year>2018</year>
          <month>06</month>
          <day>11</day>
          <access-date>2026-07-31</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://openai.com/index/language-unsupervised/">https://openai.com/index/language-unsupervised/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Brown</surname>
              <given-names>TB</given-names>
            </name>
            <name name-style="western">
              <surname>Mann</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Ryder</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Subbiah</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kaplan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Dhariwal</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Neelakantan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shyam</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Sastry</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Askell</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Agarwal</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Herbert-Voss</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Krueger</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Henighan</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Child</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ramesh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Ziegler</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Winter</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hesse</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sigler</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Litwin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gray</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chess</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Berner</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>McCandlish</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Radford</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sutskever</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Amodei</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Language models are few-shot learners</article-title>
          <source>NIPS '20: Proceedings of the 34th International Conference on Neural Information Processing Systems</source>
          <year>2020</year>
          <publisher-loc>New York, NY</publisher-loc>
          <publisher-name>Curran Associates Inc</publisher-name>
          <fpage>1877</fpage>
          <lpage>901</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Pal</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Stirrat</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Shi</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Umair</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Recent advances in artificial intelligence for radiology report generation: a brief review</article-title>
          <source>BJR Artif Intell</source>
          <year>2026</year>
          <month>01</month>
          <day>30</day>
          <volume>3</volume>
          <issue>1</issue>
          <fpage>ubag003</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/bjrai/article-lookup/doi/10.1093/bjrai/ubag003"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bjrai/ubag003</pub-id>
          <pub-id pub-id-type="medline">42063603</pub-id>
          <pub-id pub-id-type="pii">ubag003</pub-id>
          <pub-id pub-id-type="pmcid">PMC13045517</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Patil</surname>
              <given-names>NS</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>RS</given-names>
            </name>
            <name name-style="western">
              <surname>van der Pol</surname>
              <given-names>CB</given-names>
            </name>
            <name name-style="western">
              <surname>Larocque</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Comparative performance of ChatGPT and Bard in a text-based radiology knowledge assessment</article-title>
          <source>Can Assoc Radiol J</source>
          <year>2024</year>
          <month>05</month>
          <volume>75</volume>
          <issue>2</issue>
          <fpage>344</fpage>
          <lpage>50</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/08465371231193716?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/08465371231193716</pub-id>
          <pub-id pub-id-type="medline">37578849</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lecler</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Duron</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Soyer</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Revolutionizing radiology with GPT-based models: current applications, future possibilities and limitations of ChatGPT</article-title>
          <source>Diagn Interv Imaging</source>
          <year>2023</year>
          <month>06</month>
          <volume>104</volume>
          <issue>6</issue>
          <fpage>269</fpage>
          <lpage>74</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2211-5684(23)00027-X"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.diii.2023.02.003</pub-id>
          <pub-id pub-id-type="medline">36858933</pub-id>
          <pub-id pub-id-type="pii">S2211-5684(23)00027-X</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Elkassem</surname>
              <given-names>AA</given-names>
            </name>
            <name name-style="western">
              <surname>Smith</surname>
              <given-names>AD</given-names>
            </name>
          </person-group>
          <article-title>Potential use cases for ChatGPT in radiology reporting</article-title>
          <source>AJR Am J Roentgenol</source>
          <year>2023</year>
          <month>09</month>
          <volume>221</volume>
          <issue>3</issue>
          <fpage>373</fpage>
          <lpage>6</lpage>
          <pub-id pub-id-type="doi">10.2214/AJR.23.29198</pub-id>
          <pub-id pub-id-type="medline">37095665</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rao</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kamineni</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lie</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Succi</surname>
              <given-names>MD</given-names>
            </name>
          </person-group>
          <article-title>Evaluating ChatGPT as an adjunct for radiologic decision-making</article-title>
          <source>medRxiv</source>
          <comment>Preprint posted online on February 07, 2023</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1101/2023.02.02.23285399"/>
          </comment>
          <pub-id pub-id-type="doi">10.1101/2023.02.02.23285399</pub-id>
          <pub-id pub-id-type="medline">36798292</pub-id>
          <pub-id pub-id-type="pii">2023.02.02.23285399</pub-id>
          <pub-id pub-id-type="pmcid">PMC9934725</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bhayana</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Bleakney</surname>
              <given-names>RR</given-names>
            </name>
            <name name-style="western">
              <surname>Krishna</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>GPT-4 in radiology: improvements in advanced reasoning</article-title>
          <source>Radiology</source>
          <year>2023</year>
          <month>06</month>
          <volume>307</volume>
          <issue>5</issue>
          <fpage>e230987</fpage>
          <pub-id pub-id-type="doi">10.1148/radiol.230987</pub-id>
          <pub-id pub-id-type="medline">37191491</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bhayana</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Krishna</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Bleakney</surname>
              <given-names>RR</given-names>
            </name>
          </person-group>
          <article-title>Performance of ChatGPT on a radiology board-style examination: insights into current strengths and limitations</article-title>
          <source>Radiology</source>
          <year>2023</year>
          <month>06</month>
          <volume>307</volume>
          <issue>5</issue>
          <fpage>e230582</fpage>
          <pub-id pub-id-type="doi">10.1148/radiol.230582</pub-id>
          <pub-id pub-id-type="medline">37191485</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>DeepSeek-AI</collab>
          </person-group>
          <article-title>DeepSeek-V2: a strong, economical, and efficient mixture-of-experts language model</article-title>
          <source>ArXiv</source>
          <comment>Preprint posted online on May 7, 2024</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2405.04434"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2405.04434</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Xiang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Rantalainen</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>A mixture of experts (MoE) model to improve AI-based computational pathology prediction performance under variable levels of image blur</article-title>
          <source>BMC Med Imaging</source>
          <year>2025</year>
          <month>10</month>
          <day>13</day>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>407</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedimaging.biomedcentral.com/articles/10.1186/s12880-025-01974-w"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12880-025-01974-w</pub-id>
          <pub-id pub-id-type="medline">41083966</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12880-025-01974-w</pub-id>
          <pub-id pub-id-type="pmcid">PMC12516837</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hou</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Geng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Xi</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>DeepSeek R1 excels in diagnosing previously misdiagnosed cases</article-title>
          <source>Array</source>
          <year>2025</year>
          <month>12</month>
          <volume>28</volume>
          <fpage>100559</fpage>
          <pub-id pub-id-type="doi">10.1016/j.array.2025.100559</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Temsah</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Alhasan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Altamimi</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Jamal</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Al-Eyadhy</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Malki</surname>
              <given-names>KH</given-names>
            </name>
            <name name-style="western">
              <surname>Temsah</surname>
              <given-names>MH</given-names>
            </name>
          </person-group>
          <article-title>DeepSeek in healthcare: revealing opportunities and steering challenges of a new open-source artificial intelligence frontier</article-title>
          <source>Cureus</source>
          <year>2025</year>
          <month>02</month>
          <day>18</day>
          <volume>17</volume>
          <issue>2</issue>
          <fpage>e79221</fpage>
          <pub-id pub-id-type="doi">10.7759/cureus.79221</pub-id>
          <pub-id pub-id-type="medline">39974299</pub-id>
          <pub-id pub-id-type="pmcid">PMC11836063</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Roth</surname>
              <given-names>GA</given-names>
            </name>
            <name name-style="western">
              <surname>Mensah</surname>
              <given-names>GA</given-names>
            </name>
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>CO</given-names>
            </name>
            <name name-style="western">
              <surname>Addolorato</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Ammirati</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Baddour</surname>
              <given-names>LM</given-names>
            </name>
            <name name-style="western">
              <surname>Barengo</surname>
              <given-names>NC</given-names>
            </name>
            <name name-style="western">
              <surname>Beaton</surname>
              <given-names>AZ</given-names>
            </name>
            <name name-style="western">
              <surname>Benjamin</surname>
              <given-names>EJ</given-names>
            </name>
            <name name-style="western">
              <surname>Benziger</surname>
              <given-names>CP</given-names>
            </name>
            <name name-style="western">
              <surname>Bonny</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Brauer</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Brodmann</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Cahill</surname>
              <given-names>TJ</given-names>
            </name>
            <name name-style="western">
              <surname>Carapetis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Catapano</surname>
              <given-names>AL</given-names>
            </name>
            <name name-style="western">
              <surname>Chugh</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Cooper</surname>
              <given-names>LT</given-names>
            </name>
            <name name-style="western">
              <surname>Coresh</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Criqui</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>DeCleene</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Eagle</surname>
              <given-names>KA</given-names>
            </name>
            <name name-style="western">
              <surname>Emmons-Bell</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Feigin</surname>
              <given-names>VL</given-names>
            </name>
            <name name-style="western">
              <surname>Fernández-Solà</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Fowkes</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Gakidou</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Grundy</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>FJ</given-names>
            </name>
            <name name-style="western">
              <surname>Howard</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Inker</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikeyan</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Kassebaum</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Koroshetz</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Lavie</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lloyd-Jones</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>HS</given-names>
            </name>
            <name name-style="western">
              <surname>Mirijello</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Temesgen</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Mokdad</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Moran</surname>
              <given-names>AE</given-names>
            </name>
            <name name-style="western">
              <surname>Muntner</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Narula</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Neal</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Ntsekhe</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Moraes de Oliveira</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Otto</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Owolabi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pratt</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rajagopalan</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Reitsma</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ribeiro</surname>
              <given-names>AL</given-names>
            </name>
            <name name-style="western">
              <surname>Rigotti</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Rodgers</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sable</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Shakil</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sliwa-Hahnle</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Stark</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Sundström</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Timpel</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Tleyjeh</surname>
              <given-names>IM</given-names>
            </name>
            <name name-style="western">
              <surname>Valgimigli</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Vos</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Whelton</surname>
              <given-names>PK</given-names>
            </name>
            <name name-style="western">
              <surname>Yacoub</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zuhlke</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Murray</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Fuster</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Global burden of cardiovascular diseases and risk factors, 1990-2019: update from the GBD 2019 study</article-title>
          <source>J Am Coll Cardiol</source>
          <year>2020</year>
          <month>12</month>
          <day>22</day>
          <volume>76</volume>
          <issue>25</issue>
          <fpage>2982</fpage>
          <lpage>3021</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://air.unimi.it/handle/2434/809754"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jacc.2020.11.010</pub-id>
          <pub-id pub-id-type="medline">33309175</pub-id>
          <pub-id pub-id-type="pii">S0735-1097(20)37775-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC7755038</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Martin-Isla</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Campello</surname>
              <given-names>VM</given-names>
            </name>
            <name name-style="western">
              <surname>Izquierdo</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Raisi-Estabragh</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Baeßler</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Petersen</surname>
              <given-names>SE</given-names>
            </name>
            <name name-style="western">
              <surname>Lekadir</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Image-based cardiac diagnosis with machine learning: a review</article-title>
          <source>Front Cardiovasc Med</source>
          <year>2020</year>
          <month>1</month>
          <day>24</day>
          <volume>7</volume>
          <fpage>1</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/32039241"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fcvm.2020.00001</pub-id>
          <pub-id pub-id-type="medline">32039241</pub-id>
          <pub-id pub-id-type="pmcid">PMC6992607</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Perone</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Bernardi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Redheuil</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mafrica</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Conte</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Spadafora</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ecarnot</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Tokgozoglu</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Santos-Gallego</surname>
              <given-names>CG</given-names>
            </name>
            <name name-style="western">
              <surname>Kaiser</surname>
              <given-names>SE</given-names>
            </name>
            <name name-style="western">
              <surname>Fogacci</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Sabouret</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bhatt</surname>
              <given-names>DL</given-names>
            </name>
            <name name-style="western">
              <surname>Paneni</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Banach</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Santos</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Biondi Zoccai</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Ray</surname>
              <given-names>KK</given-names>
            </name>
            <name name-style="western">
              <surname>Sabouret</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Role of cardiovascular imaging in risk assessment: recent advances, gaps in evidence, and future directions</article-title>
          <source>J Clin Med</source>
          <year>2023</year>
          <month>08</month>
          <day>26</day>
          <volume>12</volume>
          <issue>17</issue>
          <fpage>5563</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=jcm12175563"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/jcm12175563</pub-id>
          <pub-id pub-id-type="medline">37685628</pub-id>
          <pub-id pub-id-type="pii">jcm12175563</pub-id>
          <pub-id pub-id-type="pmcid">PMC10487991</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sørensen</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Van den Broucke</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Fullam</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Doyle</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Pelikan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Slonska</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Brand</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Health literacy and public health: a systematic review and integration of definitions and models</article-title>
          <source>BMC Public Health</source>
          <year>2012</year>
          <month>01</month>
          <day>25</day>
          <volume>12</volume>
          <fpage>80</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcpublichealth.biomedcentral.com/articles/10.1186/1471-2458-12-80"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/1471-2458-12-80</pub-id>
          <pub-id pub-id-type="medline">22276600</pub-id>
          <pub-id pub-id-type="pii">1471-2458-12-80</pub-id>
          <pub-id pub-id-type="pmcid">PMC3292515</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Laranjo</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Dunn</surname>
              <given-names>AG</given-names>
            </name>
            <name name-style="western">
              <surname>Tong</surname>
              <given-names>HL</given-names>
            </name>
            <name name-style="western">
              <surname>Kocaballi</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bashir</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Surian</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Gallego</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Magrabi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Lau</surname>
              <given-names>AY</given-names>
            </name>
            <name name-style="western">
              <surname>Coiera</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Conversational agents in healthcare: a systematic review</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2018</year>
          <month>09</month>
          <day>01</day>
          <volume>25</volume>
          <issue>9</issue>
          <fpage>1248</fpage>
          <lpage>58</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/30010941"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocy072</pub-id>
          <pub-id pub-id-type="medline">30010941</pub-id>
          <pub-id pub-id-type="pii">5052181</pub-id>
          <pub-id pub-id-type="pmcid">PMC6118869</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bickmore</surname>
              <given-names>TW</given-names>
            </name>
            <name name-style="western">
              <surname>Trinh</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Olafsson</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>O'Leary</surname>
              <given-names>TK</given-names>
            </name>
            <name name-style="western">
              <surname>Asadi</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Rickles</surname>
              <given-names>NM</given-names>
            </name>
            <name name-style="western">
              <surname>Cruz</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Patient and consumer safety risks when using conversational assistants for medical information: an observational study of Siri, Alexa, and Google Assistant</article-title>
          <source>J Med Internet Res</source>
          <year>2018</year>
          <month>09</month>
          <day>04</day>
          <volume>20</volume>
          <issue>9</issue>
          <fpage>e11510</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2018/9/e11510/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/11510</pub-id>
          <pub-id pub-id-type="medline">30181110</pub-id>
          <pub-id pub-id-type="pii">v20i9e11510</pub-id>
          <pub-id pub-id-type="pmcid">PMC6231817</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Diviani</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>van den Putte</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Giani</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>van Weert</surname>
              <given-names>JC</given-names>
            </name>
          </person-group>
          <article-title>Low health literacy and evaluation of online health information: a systematic review of the literature</article-title>
          <source>J Med Internet Res</source>
          <year>2015</year>
          <month>05</month>
          <day>07</day>
          <volume>17</volume>
          <issue>5</issue>
          <fpage>e112</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2015/5/e112/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/jmir.4018</pub-id>
          <pub-id pub-id-type="medline">25953147</pub-id>
          <pub-id pub-id-type="pii">v17i5e112</pub-id>
          <pub-id pub-id-type="pmcid">PMC4468598</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Aydin</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Karabacak</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Vlachos</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Margetis</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Large language models in patient education: a scoping review of applications in medicine</article-title>
          <source>Front Med (Lausanne)</source>
          <year>2024</year>
          <month>10</month>
          <day>29</day>
          <volume>11</volume>
          <fpage>1477898</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/fmed.2024.1477898"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fmed.2024.1477898</pub-id>
          <pub-id pub-id-type="medline">39534227</pub-id>
          <pub-id pub-id-type="pmcid">PMC11554522</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Asgari</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Montaña-Brown</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Dubois</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Khalil</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Balloch</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yeung</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Pimenta</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>05</month>
          <day>13</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>274</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01670-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01670-7</pub-id>
          <pub-id pub-id-type="medline">40360677</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01670-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC12075489</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Langrené</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Unleashing the potential of prompt engineering for large language models</article-title>
          <source>Patterns (N Y)</source>
          <year>2025</year>
          <month>05</month>
          <day>8</day>
          <volume>6</volume>
          <issue>6</issue>
          <fpage>101260</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2666-3899(25)00108-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.patter.2025.101260</pub-id>
          <pub-id pub-id-type="medline">40575123</pub-id>
          <pub-id pub-id-type="pii">S2666-3899(25)00108-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC12191768</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>van Nuland</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lobbezoo</surname>
              <given-names>AF</given-names>
            </name>
            <name name-style="western">
              <surname>van de Garde</surname>
              <given-names>EM</given-names>
            </name>
            <name name-style="western">
              <surname>Herbrink</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>van Heijl</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Bognàr</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Houwen</surname>
              <given-names>JP</given-names>
            </name>
            <name name-style="western">
              <surname>Dekens</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wannet</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Egberts</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>van der Linden</surname>
              <given-names>PD</given-names>
            </name>
          </person-group>
          <article-title>Assessing accuracy of ChatGPT in response to questions from day to day pharmaceutical care in hospitals</article-title>
          <source>Explor Res Clin Soc Pharm</source>
          <year>2024</year>
          <month>06</month>
          <day>13</day>
          <volume>15</volume>
          <fpage>100464</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2667-2766(24)00061-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.rcsop.2024.100464</pub-id>
          <pub-id pub-id-type="medline">39050145</pub-id>
          <pub-id pub-id-type="pii">S2667-2766(24)00061-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC11267013</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Marey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Saad</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Tanas</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ghorab</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Niemierko</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Backer</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Umair</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the accuracy and reliability of AI chatbots in patient education on cardiovascular imaging: a comparative study of ChatGPT, Gemini, and Copilot</article-title>
          <source>Egypt J Radiol Nucl Med</source>
          <year>2025</year>
          <month>03</month>
          <day>27</day>
          <volume>56</volume>
          <fpage>37</fpage>
          <pub-id pub-id-type="doi">10.1186/s43055-025-01452-x</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>OpenAI</collab>
          </person-group>
          <article-title>OpenAI o1 system card</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on December 21, 2024</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2412.16720</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>OpenAI</collab>
          </person-group>
          <article-title>GPT-4o system card</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on October 25, 2024</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2410.21276</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>von Elm</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Altman</surname>
              <given-names>DG</given-names>
            </name>
            <name name-style="western">
              <surname>Egger</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pocock</surname>
              <given-names>SJ</given-names>
            </name>
            <name name-style="western">
              <surname>Gøtzsche</surname>
              <given-names>PC</given-names>
            </name>
            <name name-style="western">
              <surname>Vandenbroucke</surname>
              <given-names>JP</given-names>
            </name>
          </person-group>
          <article-title>The Strengthening the Reporting of Observational Studies in Epidemiology (STROBE) statement: guidelines for reporting observational studies</article-title>
          <source>Lancet</source>
          <year>2007</year>
          <month>10</month>
          <day>20</day>
          <volume>370</volume>
          <issue>9596</issue>
          <fpage>1453</fpage>
          <lpage>7</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://core.ac.uk/reader/33050540?utm_source=linkout"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/S0140-6736(07)61602-X</pub-id>
          <pub-id pub-id-type="medline">18064739</pub-id>
          <pub-id pub-id-type="pii">S0140-6736(07)61602-X</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="web">
          <article-title>eCFR: 45 CFR 46.102 -- definitions for purposes of this policy</article-title>
          <source>Code of Federal Regulations</source>
          <access-date>2026-06-01</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.ecfr.gov/current/title-45/subtitle-A/subchapter-A/part-46/subpart-A/section-46.102">https://www.ecfr.gov/current/title-45/subtitle-A/subchapter-A/part-46/subpart-A/section-46.102</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Scheschenja</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Viniol</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Bastian</surname>
              <given-names>MB</given-names>
            </name>
            <name name-style="western">
              <surname>Wessendorf</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>König</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Mahnken</surname>
              <given-names>AH</given-names>
            </name>
          </person-group>
          <article-title>Feasibility of GPT-3 and GPT-4 for in-depth patient education prior to interventional radiological procedures: a comparative analysis</article-title>
          <source>Cardiovasc Intervent Radiol</source>
          <year>2024</year>
          <month>02</month>
          <volume>47</volume>
          <issue>2</issue>
          <fpage>245</fpage>
          <lpage>50</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37872295"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s00270-023-03563-2</pub-id>
          <pub-id pub-id-type="medline">37872295</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00270-023-03563-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10844465</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Currie</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Robbie</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tually</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT and patient information in nuclear medicine: GPT-3.5 versus GPT-4</article-title>
          <source>J Nucl Med Technol</source>
          <year>2023</year>
          <month>12</month>
          <day>05</day>
          <volume>51</volume>
          <issue>4</issue>
          <fpage>307</fpage>
          <lpage>13</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://tech.snmjournals.org/cgi/pmidlookup?view=long&#38;pmid=37699647"/>
          </comment>
          <pub-id pub-id-type="doi">10.2967/jnmt.123.266151</pub-id>
          <pub-id pub-id-type="medline">37699647</pub-id>
          <pub-id pub-id-type="pii">jnmt.123.266151</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Omar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sorin</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>JD</given-names>
            </name>
            <name name-style="western">
              <surname>Reich</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Freeman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Gavin</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Charney</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Stump</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Bragazzi</surname>
              <given-names>NL</given-names>
            </name>
            <name name-style="western">
              <surname>Nadkarni</surname>
              <given-names>GN</given-names>
            </name>
            <name name-style="western">
              <surname>Klang</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Multi-model assurance analysis showing large language models are highly vulnerable to adversarial hallucination attacks during clinical decision support</article-title>
          <source>Commun Med (Lond)</source>
          <year>2025</year>
          <month>08</month>
          <day>02</day>
          <volume>5</volume>
          <issue>1</issue>
          <fpage>330</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s43856-025-01021-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s43856-025-01021-3</pub-id>
          <pub-id pub-id-type="medline">40753316</pub-id>
          <pub-id pub-id-type="pii">10.1038/s43856-025-01021-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC12318031</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>YB</given-names>
            </name>
            <name name-style="western">
              <surname>Ghosh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Hochberg</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Rapoport</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lallas</surname>
              <given-names>CD</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Cohen</surname>
              <given-names>SD</given-names>
            </name>
          </person-group>
          <article-title>Comparison of ChatGPT and traditional patient education materials for men's health</article-title>
          <source>Urol Pract</source>
          <year>2024</year>
          <month>01</month>
          <volume>11</volume>
          <issue>1</issue>
          <fpage>87</fpage>
          <lpage>94</lpage>
          <pub-id pub-id-type="doi">10.1097/UPJ.0000000000000490</pub-id>
          <pub-id pub-id-type="medline">37914380</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Das</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Sakib</surname>
              <given-names>SK</given-names>
            </name>
            <name name-style="western">
              <surname>Ahmed</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Trustworthy medical imaging with large language models: a study of hallucinations across modalities</article-title>
          <source>Proceedings of the 2025 IEEE/CVF International Conference on Computer Vision Workshops</source>
          <year>2025</year>
          <conf-name>ICCVW 2025</conf-name>
          <conf-date>Oct 19-20, 2025</conf-date>
          <conf-loc>Honolulu, HI</conf-loc>
          <pub-id pub-id-type="doi">10.1109/iccvw69036.2025.00136</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lewis</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Perez</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Piktus</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Petroni</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Karpukhin</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Goyal</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Küttler</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lewis</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Yih</surname>
              <given-names>WT</given-names>
            </name>
            <name name-style="western">
              <surname>Rocktäschel</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Riedel</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kiela</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Retrieval-augmented generation for knowledge-intensive NLP tasks</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on May 22, 2020</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2005.11401</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shuster</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Poff</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kiela</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Weston</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Retrieval augmentation reduces hallucination in conversation</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on April 15, 2021</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2104.07567</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zakka</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Shad</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Chaurasia</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dalal</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Moor</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Fong</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Phillips</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Alexander</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Ashley</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Boyd</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Boyd</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Hirsch</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Langlotz</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Melia</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Nelson</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sallam</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tullis</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Vogelsong</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Cunningham</surname>
              <given-names>JP</given-names>
            </name>
            <name name-style="western">
              <surname>Hiesinger</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>Almanac - retrieval-augmented language models for clinical medicine</article-title>
          <source>NEJM AI</source>
          <year>2024</year>
          <month>02</month>
          <volume>1</volume>
          <issue>2</issue>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38343631"/>
          </comment>
          <pub-id pub-id-type="doi">10.1056/aioa2300068</pub-id>
          <pub-id pub-id-type="medline">38343631</pub-id>
          <pub-id pub-id-type="pmcid">PMC10857783</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tanwani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Seneviratne</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gamble</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Babiker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schärli</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhery</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tomasev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>08</month>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>80</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref40">
        <label>40</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tian</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Mitchell</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Manning</surname>
              <given-names>CD</given-names>
            </name>
            <name name-style="western">
              <surname>Finn</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Fine-tuning language models for factuality</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on November 14, 2023</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2311.08401</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref41">
        <label>41</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Marey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mehrtabar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Afify</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pal</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Trvalik</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Adeleke</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Umair</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>From echocardiography to CT/MRI: lessons for AI implementation in cardiovascular imaging in LMICs-a systematic review and narrative synthesis</article-title>
          <source>Bioengineering (Basel)</source>
          <year>2025</year>
          <month>09</month>
          <day>27</day>
          <volume>12</volume>
          <issue>10</issue>
          <fpage>1038</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=bioengineering12101038"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/bioengineering12101038</pub-id>
          <pub-id pub-id-type="medline">41155037</pub-id>
          <pub-id pub-id-type="pii">bioengineering12101038</pub-id>
          <pub-id pub-id-type="pmcid">PMC12561239</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
