<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e98819</article-id><article-id pub-id-type="doi">10.2196/98819</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Conditional Perplexity Scoring for Large Language Model&#x2212;Generated Differential Diagnoses in Case Reports: Preliminary Computational Evaluation</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Hirosawa</surname><given-names>Takanobu</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Harada</surname><given-names>Yukinori</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kawamura</surname><given-names>Ren</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Shimizu</surname><given-names>Taro</given-names></name><degrees>MSc, MPH, MBA, MD, PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Department of Diagnostic and Generalist Medicine, Dokkyo Medical University</institution><addr-line>880 Kitakobayashi, Mibu-cho</addr-line><addr-line>Shimotsuga</addr-line><addr-line>Tochigi</addr-line><country>Japan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Lau</surname><given-names>Gabriel Rongyang</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Hu</surname><given-names>Yihan</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yu</surname><given-names>Zekai</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Takanobu Hirosawa, MD, PhD, Department of Diagnostic and Generalist Medicine, Dokkyo Medical University, 880 Kitakobayashi, Mibu-cho, Shimotsuga, Tochigi, 321-0293, Japan, 81 282-87-2498; <email>t.hirosawa1983@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>9</day><month>9</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e98819</elocation-id><history><date date-type="received"><day>19</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>18</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>19</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Takanobu Hirosawa, Yukinori Harada, Ren Kawamura, Taro Shimizu. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 9.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e98819"/><abstract><sec><title>Background</title><p>Large language models (LLMs) are increasingly used to generate differential diagnoses from clinical narratives. However, LLM-based diagnostic clinical decision support systems still lack a quantitative measure of how strongly a diagnosis is supported by the available case description. Conditional perplexity score quantifies how predictable a target text is given in a preceding context, with lower scores indicating greater predictability. We hypothesized that this concept can be adapted to diagnostic reasoning by treating the prediagnostic case description as the context and a diagnosis as the target text.</p></sec><sec><title>Objective</title><p>This study aims to evaluate whether conditional perplexity scores, computed by an independent LLM and conditioned on case-report narratives, differ between physician-verified correct and incorrect LLM-generated diagnoses. Specifically, we hypothesized that the correct LLM-generated diagnosis verified by physicians would have lower conditional perplexity scores than incorrect LLM-generated differential diagnoses. A secondary outcome was to compare this scoring behavior across differential diagnosis lists generated by different LLMs.</p></sec><sec sec-type="methods"><title>Methods</title><p>We performed a preliminary computational analysis of 392 peer-reviewed diagnostic case reports published in the <italic>American Journal of Case Reports</italic> in 2022. For each case, the prediagnostic clinical description was used as the conditioning context, and the case report&#x2013;defined final diagnoses were treated as the gold standard. Conditional perplexity scores for differential diagnosis lists previously generated by LLaMA2, Bard, and GPT-4 were computed using an independent longer-context LLM, Qwen2.5&#x2010;1.5B. We compared case report&#x2013;defined final diagnoses, correct LLM-generated diagnoses verified by physicians, and incorrect generated diagnoses using nonparametric comparisons and receiver operating characteristic analyses.</p></sec><sec sec-type="results"><title>Results</title><p>All 392 cases had complete case descriptions and case report&#x2013;defined final diagnoses. Across the top-10 differential diagnosis lists generated by LLaMA2, Bard, and GPT-4, 823 correct LLM-generated diagnoses verified by physicians and 10,875 incorrect generated diagnoses were analyzed. Case report&#x2013;defined final diagnoses had lower conditional perplexity scores than incorrect generated diagnoses (median 39.9, IQR 17.7&#x2010;119.9 vs median 133.3, IQR 37.5&#x2010;672.1). Correct LLM-generated diagnoses also had lower conditional perplexity scores than incorrect LLM-generated diagnoses (median 43.3, IQR 16.6&#x2010;147.5 vs median 133.3, IQR 37.6&#x2010;672.1). Candidate-level discrimination was moderate overall (area under the receiver operating characteristic curve [AUC] 0.666, 95% CI 0.644&#x2010;0.689) and was the highest for GPT-4&#x2013;generated differential diagnosis lists (AUC 0.678, 95% CI 0.652&#x2010;0.705), followed by LLaMA2 (AUC 0.662, 95% CI 0.625&#x2010;0.698) and Bard (AUC 0.648, 95% CI 0.617&#x2010;0.681). In within-case analyses, correct diagnoses had lower conditional perplexity than the mean incorrect diagnosis in 88.1% (237/269) to 91.1% (195/214) of evaluable lists.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Conditional perplexity provided a moderate quantitative signal associated with physician-verified correctness but did not reliably rank the correct diagnosis ahead of the strongest incorrect candidate, limiting its use as a stand-alone reranking method.</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>diagnosis</kwd><kwd>generative artificial intelligence</kwd><kwd>large language model</kwd><kwd>natural language processing</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Diagnostic Excellence and the Challenge of Differential Diagnosis</title><p>Accurate diagnosis is a central task in clinical reasoning and a major determinant of patient safety, timely treatment, and efficient access to health care resources [<xref ref-type="bibr" rid="ref1">1</xref>]. Diagnostic error remains an important cause of preventable harm, and improving diagnostic performance has therefore become a priority in contemporary health care quality and safety efforts [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. In this context, the concept of diagnostic excellence emphasizes not only diagnostic accuracy but also timeliness, communication, equity, patient-centeredness, and appropriate use of available data and expertise during the diagnostic process [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Tools that help clinicians generate, compare, and revisit diagnostic hypotheses may therefore contribute to diagnostic excellence when they are thoughtfully integrated into real-world clinical workflows [<xref ref-type="bibr" rid="ref11">11</xref>].</p></sec><sec id="s1-2"><title>Diagnostic Clinical Decision Support Systems</title><p>Clinical decision support systems (CDSSs) have been proposed as one approach to achieving diagnostic excellence [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. CDSSs are commonly classified as knowledge-based or non&#x2013;knowledge-based systems. Knowledge-based CDSSs, including rule-based systems and Bayesian models, rely on explicitly encoded clinical knowledge to support clinicians in structuring problem representations and reducing the omission of relevant differential diagnoses. However, many traditional systems require structured inputs, rely on manually curated rules, or show limited flexibility when confronted with the narrative, ambiguous, and evolving nature of real clinical information [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref16">16</xref>]. As electronic clinical text and biomedical knowledge have expanded, interest has grown in computational approaches capable of operating directly on free-text clinical information, physical examination findings, laboratory results, and imaging summaries without manual feature engineering [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>].</p></sec><sec id="s1-3"><title>Large Language Models as Emerging Diagnostic Support Systems</title><p>Recent large language models (LLMs) have renewed interest in diagnostic CDSSs because they can process clinical narratives and generate differential diagnoses using natural language processing techniques [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>]. This capability has prompted interest in their potential role as supportive tools, particularly for hypothesis generation, prioritization of rare diseases, and medical education [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. At the same time, the clinical use of LLMs raises important concerns regarding reliability, calibration, transparency, and explainability [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>]. Model outputs may appear fluent and convincing even when the underlying reasoning is incomplete, unsupported by the case details, or influenced by spurious textual associations [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Consequently, established methods are needed not only to assess whether an LLM-generated differential diagnosis list can include the final diagnosis but also to quantify how strongly a differential diagnosis is supported by the preceding clinical narrative.</p></sec><sec id="s1-4"><title>Limitations of Current Evaluation Approaches</title><p>Many prior evaluations of LLM-based diagnostic performance have focused on whether the final diagnosis appears anywhere in a model-generated list or on its ordinal rank within a differential diagnosis list [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref33">33</xref>]. These measures are clinically intuitive and remain useful for benchmarking diagnostic CDSS performance. To move beyond simple accuracy and evaluate the quality of the reasoning process, recent studies have increasingly adopted clinical reasoning rubrics such as the Revised-IDEA (R-IDEA) score. The R-IDEA framework uses a 10-point scale to grade 4 core domains of clinical reasoning: the interpretive summary, the differential diagnosis, the explanation of the leading diagnosis, and the justification for alternative diagnoses [<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>However, these approaches do not directly quantify the degree of compatibility between narrative clinical information and a specific diagnosis. A diagnosis could be placed in a top-<italic>k</italic> list, meaning it is among the first <italic>k</italic> differential diagnoses proposed by a model, while still being only weakly supported by the available clinical information [<xref ref-type="bibr" rid="ref35">35</xref>]. In contrast, a diagnosis that is highly consistent with the narrative clinical information might receive stronger model-based support even if it was generated by a different model or phrased differently. A complementary quantitative framework is therefore needed to evaluate narrative-to-diagnosis compatibility directly.</p></sec><sec id="s1-5"><title>Explainable AI and the Need for Quantitative Compatibility Metrics</title><p>Such a framework is particularly relevant for the development of explainable artificial intelligence (XAI) in medicine [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. In high-stakes clinical settings, explainability extends beyond producing plausible reasoning; it also involves providing transparent, reproducible signals that help users understand why one diagnostic hypothesis is more strongly supported than another. From this perspective, a useful CDSS metric should be externally computable, comparable across differential diagnosis lists, and interpretable as a function of the observed case narrative rather than an internal score produced by a single LLM. The quantitative measures may therefore help make LLM-assisted differential diagnosis more inspectable and auditable [<xref ref-type="bibr" rid="ref38">38</xref>].</p></sec><sec id="s1-6"><title>LLM-Based Conditional Perplexity Score as a Compatibility Measure</title><p>Perplexity score is a standard NLP metric used to evaluate how well an LLM predicts a sequence of tokens [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. A token is the basic unit of data that an LLM processes and generates. In an LLM, the conditional perplexity score of a target sequence can be estimated token by token, given the preceding text as context [<xref ref-type="bibr" rid="ref41">41</xref>]. Applied to clinical diagnosis, a lower conditional negative log-likelihood (NLL) and a correspondingly lower conditional perplexity score indicate that a specific diagnosis string is more predictable to the scoring model after it has processed the case narrative. In this context, the conditional perplexity score can be repurposed as a reproducible measure of text-based compatibility between the available clinical narrative and a proposed differential diagnosis.</p><p>This approach offers several potential advantages for both CDSS research and the evaluation of XAI in medicine. First, it provides a continuous score rather than binary correct or incorrect judgments or rubric-based scores. Second, it facilitates a direct comparison among multiple differential diagnoses generated by different LLMs. Third, because the score is conditioned explicitly on the prediagnostic narrative, it offers a transparent way to assess whether corrected LLM-generated diagnoses, verified by physicians, exhibit higher text-based compatibility than incorrect differential diagnoses. Fourth, such scores may serve as a reranking signal within ensemble CDSS approaches [<xref ref-type="bibr" rid="ref42">42</xref>], in which differential diagnoses generated by multiple models can be aggregated, scored for narrative compatibility, and reordered to surface the most plausible clinical conclusion.</p></sec><sec id="s1-7"><title>Remaining Knowledge Gaps</title><p>Despite growing interest in LLM-assisted diagnostic reasoning, several important questions remain insufficiently addressed in the existing literature. First, it is unclear whether LLM-based compatibility metrics can reliably distinguish correct LLM-generated diagnoses verified by physicians from incorrect LLM-generated differential diagnoses when conditioned on the same case descriptions. Second, prior work has rarely examined whether externally computed LLM-based scores align with correct LLM-generated diagnoses verified by physicians within prediagnostic case descriptions. Third, the extent to which such scores could function as a reproducible reranking signal across differential diagnosis lists generated by different LLM families has not been systematically evaluated. Addressing these gaps is important for determining whether LLM-based metrics can contribute to explainable evaluation and safer clinical decision support.</p></sec><sec id="s1-8"><title>Study Objective and Research Questions</title><p>In the present study, we evaluated 392 diagnostic case reports published in 2022 in the <italic>American Journal of Case Reports</italic>. Using precompiled differential diagnosis lists generated by LLaMA2, Bard, and GPT-4, we applied a longer-context causal language model as an external scorer to compute conditional perplexity scores for the gold-standard final diagnoses and differential diagnosis lists.</p><p>Our primary objective was to determine whether external LLM-based conditional perplexity scores could serve as a quantitative framework for ranking physician-verified correct LLM-generated diagnoses within differential diagnosis lists. The specified aim was to test whether correct diagnoses would receive lower scores than incorrect candidate diagnoses when conditioned on the same prediagnostic case narrative. We further aimed to determine whether this signal remained observable across multiple source LLMs while explicitly treating the score as a text-level compatibility measure rather than a stand-alone measure of clinical reasoning.</p><list list-type="order"><list-item><p>Compatibility: Do case report&#x2013;defined final diagnoses receive lower conditional perplexity scores than incorrect LLM-generated differential diagnoses when conditioned on the same prediagnostic case narrative?</p></list-item><list-item><p>Candidate discrimination: Within LLM-generated differential diagnosis lists, can a conditional perplexity score distinguish physician-verified correct LLM-generated diagnoses from incorrect differential diagnoses?</p></list-item><list-item><p>Cross-model behavior: Does this compatibility signal remain observable across differential diagnosis lists generated by multiple LLMs?</p></list-item></list><p>By framing diagnostic evaluation through an LLM-based conditional perplexity score, this study explores a preliminary approach for connecting LLM-based differential diagnosis generation with reproducible evaluation, interpretable decision support, and the broader goal of diagnostic excellence.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Setting</title><p>This was a retrospective computational evaluation of 392 diagnostic case reports published in 2022 in the <italic>American Journal of Case Reports</italic>. The objective was not to generate new diagnoses but to quantify how compatible each diagnosis was with the preceding clinical narrative using a fixed external LLM. The overall study flow is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study workflow. Case reports published in 2022 in the <italic>American Journal of Case Reports</italic> were collected, prediagnostic case descriptions were extracted, case report&#x2013;defined final diagnoses were recorded, and large language model (LLM)&#x2013;generated differential diagnosis lists from GPT-4, Bard, and LLaMA2 were rescored by a longer-context LLM to compute conditional perplexity scores.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e98819_fig01.png"/></fig></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study was a secondary computational analysis of previously published, publicly available case reports and previously generated differential diagnosis lists. The investigators had no contact with patients and did not access medical records, direct identifiers, or nonpublic health information. Because the study used only previously published, publicly available information and did not involve identifiable private information or direct interaction with human participants, additional ethics committee review and informed consent were not sought. Therefore, additional ethics committee review and informed consent were not sought. The source case reports were accessed under the publication and licensing terms of th<italic>e American Journal of Case Reports</italic> and were used for noncommercial computational research. The present study did not redistribute modified full-text case reports. This retrospective computational analysis was not preregistered.</p></sec><sec id="s2-3"><title>Data Preprocessing</title><p>Clinical information was extracted from the original case reports prior to the final diagnosis. The prediagnostic case descriptions were prepared by the main investigator (TH) after removing the assessment and differential diagnoses. Typical prediagnostic case descriptions included the background, chief concerns, history of present illness, past medical history, physical examination findings, and the results of investigations. This process was validated by another investigator (YH).</p><p>For each case, the case report&#x2013;defined final diagnosis was treated as the gold standard. Duplicate diagnoses were removed after lowercasing while preserving the original order. Physician-verification labels for the LLM-generated differential diagnoses were imported from the source dataset publication. In that source dataset, 2 general internal medicine expert physicians independently reviewed each case report&#x2013;defined final diagnosis and each AI-generated differential diagnosis list. Each candidate was coded as correct if it accurately matched the final diagnosis with acceptable specificity or was sufficiently close such that appropriate treatment would be initiated without compromising patient safety; candidates judged substantially different from the final diagnosis were coded as incorrect. Disagreements were resolved through consultation with a third general internal medicine expert physician, reviewers were blinded to the AI system that produced each list, and interrater agreement was reported as 88.9% with a Cohen &#x03BA; coefficient of 0.76 [<xref ref-type="bibr" rid="ref43">43</xref>]. In the present analysis, physician-verified correctness was based on clinical concordance rather than exact string matching; therefore, clinically equivalent synonyms or differences in diagnostic granularity could be judged correct.</p><p>For each case narrative, a standardized prompt was constructed as follows: &#x201C;Clinical information from a case report: [Insert pre-diagnostic case description]. The final diagnosis is.&#x201D; The prompt was included in the conditioning context, allowing the analysis to estimate the conditional likelihood of the diagnosis tokens alone.</p></sec><sec id="s2-4"><title>Source Differential Diagnosis Lists and Model Provenance</title><p>The differential diagnosis lists were generated in a previously published Digital Health study using public web-based chat interfaces [<xref ref-type="bibr" rid="ref43">43</xref>]. The GPT-4 differential diagnosis lists were generated through the ChatGPT web interface from June 22 to June 29, 2023, using the March 24 version in the default chat mode; custom instructions were not used, and chat history was disabled in the browser used for the study. The Bard differential diagnosis lists were generated through the Google Bard web interface, now Google Gemini, from July 27 to August 1, 2023, before the system transition to Google Gemini; no specific version number or adjustable decoding settings were available, and Bard or Gemini activity was disabled. The LLaMA2 differential diagnosis lists were generated through the public LLaMA2 chatbot web interface from August 3 to August 8, 2023, using the 70B version of the LLaMA2 model by Meta AI; the web interface displayed a temperature of 2.49, top P 0.50, a maximum sequence length of 2048, and default prompt settings. The present study did not regenerate these differential diagnosis lists but rescored the fixed lists with Qwen2.5&#x2010;1.5B.</p></sec><sec id="s2-5"><title>Scoring an LLM</title><p>An LLM with an extended context length, Qwen2.5&#x2010;1.5B, was used as the scoring model. Specifically, we used the base checkpoint available from the Hugging Face repository Qwen or Qwen2.5&#x2010;1.5B. Because Qwen2.5 supports longer context windows than many earlier open-weight models, it was suitable for long case descriptions [<xref ref-type="bibr" rid="ref44">44</xref>]. Across all main analytic prompt-diagnosis sequences, all sequences fit within the 32,768-token context window; therefore, sliding-window scoring was not required.</p><p>The tokenizer was loaded from the same Hugging Face repository using AutoTokenizer with the fast tokenizer enabled (use_fast=True), and both the conditioning prompt and target diagnosis were tokenized with special-token insertion disabled (add_special_tokens=False). The model was loaded with the Hugging Face transformers library, and inference was performed on graphics processing unit (GPU) or central processing unit (CPU) depending on hardware availability [<xref ref-type="bibr" rid="ref45">45</xref>]. Because perplexity scores are model-dependent, the use of one general-purpose scorer was treated as a source of uncertainty rather than as evidence that the observed signal would necessarily reproduce across other scorer architectures or biomedical-domain models.</p></sec><sec id="s2-6"><title>Conditional Likelihood and Perplexity Calculation</title><p>For a clinical context string (<italic>c</italic>) and a candidate diagnosis string (<italic>d</italic>=<italic>x</italic><sub>1:N</sub>), where (<italic>x</italic><sub><italic>i</italic></sub>) denotes the (<italic>i</italic>)-th diagnosis token and (<italic>N</italic>) denotes the number of diagnosis tokens after tokenization by the scoring model, the conditional probability assigned to the diagnosis by a causal language model with parameters (theta) is defined as follows:</p><disp-formula><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:munderover><mml:mo>&#x220F;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mo>&#x003C;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>The token-level average NLL for the diagnosis, given the clinical context was calculated as:</p><disp-formula><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>avgNLL</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mo>&#x003C;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>The conditional perplexity score was then defined as the exponential of the average negative log-likelihood:</p><disp-formula><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="normal">P</mml:mi><mml:mi mathvariant="normal">P</mml:mi><mml:mi mathvariant="normal">L</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">g</mml:mi><mml:mi mathvariant="normal">N</mml:mi><mml:mi mathvariant="normal">L</mml:mi><mml:mi mathvariant="normal">L</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2223;</mml:mo><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mo>&#x003C;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>A lower average NLL and a lower conditional perplexity score indicate that the scoring model considered the diagnosis string more predictable after inputting the case description. When the concatenated prompt and diagnosis fit within the model context window, the entire sequence was scored in a single forward pass, with labels for prompt tokens masked out [<xref ref-type="bibr" rid="ref46">46</xref>].</p></sec><sec id="s2-7"><title>Candidate Ranking Procedure</title><p>For each case and each LLM, the 10 differential diagnoses were scored independently using the same conditioned prompt derived from prediagnostic clinical descriptions. Differential diagnoses were then sorted in ascending order of the average NLL. The diagnosis with the lowest average NLL was considered the top-ranked diagnosis according to the scoring model.</p></sec><sec id="s2-8"><title>Outcomes</title><p>The primary outcome was whether clinically supported diagnoses received lower conditional perplexity than unsupported diagnoses. Clinically supported diagnoses were analyzed as (1) the case report&#x2013;defined final diagnosis (gold standard) and (2) correct LLM-generated diagnoses verified by physicians, present within LLaMA2-generated, Bard-generated, or GPT-4&#x2013;generated differential diagnosis lists. The incorrect generated differential diagnoses served as the comparison group. Because conditional perplexity can be affected by lexical frequency, diagnosis-string length, overlap between diagnosis terms and the narrative, and scorer-model familiarity, lower scores were interpreted as greater conditional predictability under the scoring model rather than as direct evidence of case-specific clinical compatibility or clinical reasoning.</p><p>Secondary outcomes included the discriminative performance of the conditional perplexity score at the candidate level, quantified by the area under the receiver operating characteristic curve (AUC). Additional secondary outcomes included comparisons of conditional perplexity scoring behavior across LLaMA2, Bard, and GPT-4 and within-case paired evaluations. Within cases containing a physician-verified correct diagnosis, the conditional perplexity of the correct diagnosis was compared with the mean, median, and lowest conditional perplexity among incorrect diagnoses from the same differential diagnosis list. The lowest-perplexity incorrect diagnosis represented the most stringent within-list comparator.</p></sec><sec id="s2-9"><title>Statistical Analysis</title><p>Continuous variables are summarized as medians with IQRs. Receiver operating characteristic (ROC) analyses were performed at the candidate level to assess how well conditional perplexity discriminated physician-verified correct LLM-generated diagnoses from incorrect LLM-generated diagnoses. Because multiple candidate diagnoses were nested within cases, 95% CIs for the area under the ROC curve (AUC) were estimated using case-cluster nonparametric bootstrap resampling, in which cases were sampled with replacement, and all candidate diagnoses belonging to each sampled case were retained [<xref ref-type="bibr" rid="ref47">47</xref>]. Thus, the primary inferential emphasis was placed on case-cluster bootstrap estimates of candidate-level AUCs and within-case paired comparisons. Within-case comparisons of conditional perplexity scores were performed using 2-sided Wilcoxon signed-rank tests. Ninety-five percent CIs for medians reported in the within-case analyses were estimated using a nonparametric percentile bootstrap with 2000 resamples. All tests were 2-sided, and <italic>P</italic>&#x003C;.05 was considered statistically significant.</p><p>Post hoc sensitivity analyses were performed to address reviewer concerns about lexical confounding and clustering. We derived lexical features for each candidate diagnosis, including diagnosis-string token count, character count, a candidate-corpus token-frequency proxy (mean log token document frequency), token-overlap fraction between the diagnosis string and prediagnostic case description, any token overlap, exact phrase occurrence in the case description, and case description word count. We compared conditional perplexity with lexical-overlap baselines using case-level cluster bootstrap resampling by case. We also fitted an exploratory logistic regression model predicting physician-verified correctness from average NLL, diagnosis length, token-frequency proxy, lexical-overlap fraction, exact phrase occurrence, case length, and source LLM, with SEs clustered by case. These analyses were exploratory because they used a corpus-derived lexical-frequency proxy rather than the scorer model&#x2019;s true pretraining-token frequency and did not include rescoring against mismatched case descriptions.</p><p>Analyses were performed in Python (version 3.12.3). Sample Python code to calculate conditional perplexity scores and the computational environment are shown in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Additionally, sample Python code for statistical analysis and post hoc sensitivity analyses, including lexical-overlap features, adjusted logistic regression, case-level cluster-bootstrap AUCs, and within-case comparator analyses, is shown in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Dataset Characteristics and Diagnostic Coverage</title><p>The final analysis computed conditional perplexity scores for all 392 case report&#x2013;defined final diagnoses and for 11,698 generated differential diagnoses across the 3 source LLMs, including LLaMA2, Bard, and GPT-4. Overall, the generated lists contained 823 correct LLM-generated diagnoses verified by physicians and 10,875 incorrect differential diagnoses.</p><p>Correct diagnosis coverage differed across source models. Of the total 392 cases, LLaMA2 included a correct diagnosis in 214 (54.6%) of the cases, Bard in 269 (68.6%) of the cases, and GPT-4 in 340 (86.7%) of the cases. GPT-4 also achieved the best original-list top-1 coverage (n=214, 54.6%), followed by Bard (n=123, 31.4%) and LLaMA2 (n=90, 23.0%; <xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Diagnostic coverage of correct large language model (LLM)&#x2013;generated diagnoses verified by physicians within differential diagnosis lists.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Top-10, n (%)</td><td align="left" valign="bottom">Top-1, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">LLaMA2</td><td align="left" valign="top">214 (54.6)</td><td align="left" valign="top">90 (23.0)</td></tr><tr><td align="left" valign="top">Bard</td><td align="left" valign="top">269 (68.6)</td><td align="left" valign="top">123 (31.4)</td></tr><tr><td align="left" valign="top">GPT-4</td><td align="left" valign="top">340 (86.7)</td><td align="left" valign="top">214 (54.6)</td></tr></tbody></table></table-wrap><p>In <xref ref-type="table" rid="table2">Table 2</xref>, we selected 2 representative examples from the dataset: one in which the physician-verified final diagnosis exactly matched the correct LLM-generated diagnosis string, and one in which the correct LLM-generated diagnosis was clinically concordant but not text-identical to the case report&#x2013;defined diagnosis. In both examples, lower conditional perplexity indicates greater compatibility between the prediagnostic case narrative and the candidate diagnosis.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Representative examples with conditional perplexity score.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Case title</td><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Case report&#x2013;defined final diagnosis as gold standard (conditional perplexity score)</td><td align="left" valign="bottom">Correct LLM<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>-generated diagnosis (conditional perplexity score)</td><td align="left" valign="bottom">Incorrect LLM-generated diagnosis (conditional perplexity score)<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">The many faces of immune checkpoint inhibitor-associated pneumonitis: 4 case reports</td><td align="left" valign="top">Bard</td><td align="left" valign="top">Immune checkpoint inhibitor-associated pneumonitis (19.0)</td><td align="left" valign="top">Immune checkpoint inhibitor-associated pneumonitis (19.0)</td><td align="left" valign="top">Infection (30.7)</td></tr><tr><td align="left" valign="top">Solitary soft-tissue metastasis of a pancreatic adenocarcinoma 2 years after curative resection</td><td align="left" valign="top">GPT-4</td><td align="left" valign="top">Soft tissue metastasis from pancreatic adenocarcinoma (12.2)</td><td align="left" valign="top">Metastatic pancreatic adenocarcinoma (9.0)</td><td align="left" valign="top">Fibroma or myofibroma (148.5)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table2fn2"><p><sup>b</sup>Incorrect large language model&#x2013;generated diagnosis was selected based on the lowest conditional perplexity score within the differential diagnosis list.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Overall Separation of Clinically Supported Diagnoses From Incorrect Diagnoses</title><p>Case report&#x2013;defined final diagnoses had substantially lower conditional perplexity scores than incorrect LLM-generated differential diagnoses (median 39.9, IQR 17.7&#x2010;119.9 vs median 133.3, IQR 37.6&#x2010;672.1; <italic>P</italic>&#x003C;.001). Correct LLM-generated diagnoses verified by physicians showed a similar pattern, with lower conditional perplexity score than incorrect LLM-generated differential diagnoses overall (median 43.3, IQR 16.6&#x2010;147.5; <italic>P</italic>&#x003C;.001). The distributions of conditional perplexity scores across diagnosis categories are shown descriptively in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Conditional perplexity distributions show pooled score distributions for case report&#x2013;defined final diagnoses (gold final dx), correct large language model (LLM)&#x2013;generated diagnoses verified by physicians (correct generated), and incorrect LLM-generated differential diagnoses (incorrect generated) on a logarithmic <italic>y</italic>-axis.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e98819_fig02.png"/></fig><p>Using case-cluster bootstrap resampling to account for candidate diagnoses nested within cases, the pooled AUC for distinguishing physician-verified correct from incorrect generated diagnoses was 0.666 (95% CI 0.644&#x2010;0.689; <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). This AUC should be interpreted as exploratory discrimination within a curated case-report dataset rather than as a validated clinical decision threshold (<xref ref-type="table" rid="table3">Table 3</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Conditional perplexity scores according to diagnosis category.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Group</td><td align="left" valign="bottom">Diagnostic case reports, n</td><td align="left" valign="bottom">Median conditional perplexity score (IQR)</td><td align="left" valign="bottom">AUC<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> vs incorrect (case-cluster bootstrap 95% CI)<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Case report&#x2013;defined final diagnosis</td><td align="left" valign="top">392</td><td align="left" valign="top">39.9 (17.7&#x2010;119.9)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Correct LLM<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup>-generated diagnosis verified by physicians (pooled)</td><td align="left" valign="top">823</td><td align="left" valign="top">43.3 (16.6&#x2010;147.5)</td><td align="left" valign="top">0.666 (0.644&#x2010;0.689)</td></tr><tr><td align="left" valign="top">Incorrect generated diagnosis (pooled)</td><td align="left" valign="top">10,875</td><td align="left" valign="top">133.3 (37.6&#x2010;672.1)</td><td align="left" valign="top">Reference</td></tr><tr><td align="left" valign="top">LLaMA2: correct generated diagnosis</td><td align="left" valign="top">214</td><td align="left" valign="top">45.2 (17.8&#x2010;167.1)</td><td align="left" valign="top">0.662 (0.625&#x2010;0.698)</td></tr><tr><td align="left" valign="top">LLaMA2: incorrect generated diagnosis</td><td align="left" valign="top">3682</td><td align="left" valign="top">131.3 (38.0&#x2010;674.1)</td><td align="left" valign="top">Reference</td></tr><tr><td align="left" valign="top">Bard: correct generated diagnosis</td><td align="left" valign="top">269</td><td align="left" valign="top">55.9 (25.8&#x2010;215.3)</td><td align="left" valign="top">0.648 (0.617&#x2010;0.681)</td></tr><tr><td align="left" valign="top">Bard: incorrect generated diagnosis</td><td align="left" valign="top">3630</td><td align="left" valign="top">174.9 (48.9&#x2010;985.1)</td><td align="left" valign="top">Reference</td></tr><tr><td align="left" valign="top">GPT-4: correct generated diagnosis</td><td align="left" valign="top">340</td><td align="left" valign="top">30.7 (11.5&#x2010;91.8)</td><td align="left" valign="top">0.678 (0.652&#x2010;0.705)</td></tr><tr><td align="left" valign="top">GPT-4: incorrect generated diagnosis</td><td align="left" valign="top">3563</td><td align="left" valign="top">102.9 (29.0&#x2010;462.2)</td><td align="left" valign="top">Reference</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table3fn2"><p><sup>b</sup>AUCs quantify candidate-level discrimination; 95% CIs were estimated by bootstrap resampling at the case level, retaining all candidate diagnoses within each sampled case.</p></fn><fn id="table3fn3"><p><sup>c</sup>Not applicable.</p></fn><fn id="table3fn4"><p><sup>d</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Post Hoc Lexical-Overlap and Cluster-Bootstrap Sensitivity Analyses</title><p>Post hoc sensitivity analyses are summarized in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. Lexical overlap alone showed weaker discrimination than conditional perplexity: the pooled AUC was 0.570 (95% CI 0.545&#x2010;0.596) for lexical-overlap fraction, 0.567 (95% CI 0.542&#x2010;0.589) for any lexical overlap, and 0.500 (95% CI 0.485&#x2010;0.515) for exact phrase occurrence in the case description. In the adjusted logistic regression model with case-clustered SEs, lower average NLL, represented as higher-average NLL, remained associated with physician-verified correctness after adjustment for diagnosis length, token-frequency proxy, lexical overlap, exact phrase occurrence, case length, and source LLM (odds ratio 1.36 per 1-unit increase in average NLL, 95% CI 1.27&#x2010;1.46; <italic>P</italic>&#x003C;.001). The combined conditional-perplexity plus lexical-feature model had a cluster-bootstrap AUC of 0.694 (95% CI 0.672&#x2010;0.716), compared with 0.639 (95% CI 0.618&#x2010;0.662) for the lexical-only model and 0.671 (95% CI 0.649&#x2010;0.694) for the conditional-perplexity-only model.</p></sec><sec id="s3-4"><title>Model-Specific Scoring Behavior</title><p>Within each LLM as shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>, physician-verified correct LLM-generated diagnoses had a lower median conditional perplexity score than incorrect LLM-generated differential diagnoses: LLaMA2 (45.2 vs 131.3), Bard (55.9 vs 174.9), and GPT-4 (30.7 vs 102.9). Candidate-level discrimination was the highest for GPT-4 (AUC 0.678, 95% CI 0.647&#x2010;0.707), followed by LLaMA2 (AUC 0.662, 95% CI 0.626&#x2010;0.696) and Bard (AUC 0.648, 95% CI 0.616&#x2010;0.682).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Candidate-level receiver operating characteristic curves for conditional perplexity scores, shown for each source large language model (LLM) and the pooled candidate set. Because candidate diagnoses were nested within cases, uncertainty around area under the receiver operating characteristic curve (AUC) estimates was quantified using case-cluster bootstrap resampling.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e98819_fig03.png"/></fig><p>Correct GPT-4&#x2013;generated diagnoses had lower conditional perplexity score than those from LLaMA2 and Bard, whereas the difference between correct diagnoses generated by LLaMA2 and Bard was not statistically significant. False-positive conditional perplexity scores also differed across LLMs (<italic>P</italic>&#x003C;.001), with Bard showing the highest median values and GPT-4 showing the lowest median values.</p></sec><sec id="s3-5"><title>Within-Case Paired Comparisons</title><p>In paired case-level analyses restricted to cases in which a correct LLM-generated diagnosis verified by physicians was present, the correct diagnosis had a lower conditional perplexity score than the average incorrect diagnosis in 195 out of 214 (91.1%) LLaMA2 cases, 304 out of 340 (89.4%) GPT-4 cases, and 237 out of 269 (88.1%) Bard cases (all Wilcoxon <italic>P</italic>&#x003C;.001; <xref ref-type="table" rid="table4">Table 4</xref>). This comparison summarizes separation from the average incorrect candidate and should not be interpreted as indicating that the correct diagnosis was the lowest-perplexity candidate within a list.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Within-case comparison of correct generated diagnoses versus the mean conditional perplexity of incorrect generated diagnoses.<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Source</td><td align="left" valign="bottom">Paired cases, n</td><td align="left" valign="bottom">Correct diagnosis median conditional perplexity score (IQR; 95% CI)</td><td align="left" valign="bottom">Average incorrect diagnosis median conditional perplexity score (IQR; 95% CI)</td><td align="left" valign="bottom">Correct &#x003C; average incorrect, n (%)</td><td align="left" valign="bottom"><italic>P</italic> value<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Case report&#x2013;defined final diagnoses</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">39.9 (17.7-119.9; 33.9&#x2010;43.9)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">LLaMA2</td><td align="left" valign="top">214</td><td align="left" valign="top">45.1 (17.8-167.1; 36.3&#x2010;52.5)</td><td align="left" valign="top">973.5 (292.2-3980.3; 774.3&#x2010;1237.3)</td><td align="left" valign="top">195 (91.1)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Bard</td><td align="left" valign="top">269</td><td align="left" valign="top">55.9 (25.8-215.3; 44.8&#x2010;74.7)</td><td align="left" valign="top">1600.3 (443.3-6648.5; 1320.9&#x2010;2067.0)</td><td align="left" valign="top">237 (88.1)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">GPT-4</td><td align="left" valign="top">340</td><td align="left" valign="top">30.7 (11.5-91.7; 26.7&#x2010;39.1)</td><td align="left" valign="top">692.2 (243.8-2073.8; 555.2&#x2010;839.3)</td><td align="left" valign="top">304 (89.4)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Median 95% CIs were estimated using a nonparametric percentile bootstrap with 2000 resamples.</p></fn><fn id="table4fn2"><p><sup>b</sup><italic>P</italic> values are derived from 2-sided Wilcoxon signed-rank tests. </p></fn><fn id="table4fn3"><p><sup>c</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>More stringent within-case analyses showed that the physician-verified correct diagnosis had lower conditional perplexity than the lowest-perplexity incorrect diagnosis in 54 of 214 (25.2%) LLaMA2 lists, 64 of 269 (23.8%) Bard lists, and 82 of 340 (24.1%) GPT-4 lists. Compared with the median incorrect diagnosis, the corresponding proportions were 70.1% (150/214) for LLaMA2, 68.0% (183/269) for Bard, and 74.1% (252/340) for GPT-4. These findings indicate that although correct diagnoses tended to have lower perplexity than the overall distribution of incorrect diagnoses, conditional perplexity did not reliably assign the lowest score to the correct diagnosis within individual lists (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p><p>Compared with the case report&#x2013;defined final diagnosis, the corresponding correct LLM-generated diagnosis verified by physicians had a significantly higher conditional perplexity score for Bard (paired median 39.9 vs 55.9; 2-sided Wilcoxon <italic>P</italic>&#x003C;.001). However, for both LLaMA2 and GPT-4, the conditional perplexity scores for the correct generated diagnoses were comparable to the case report&#x2013;defined final diagnoses (LLaMA2 paired median 39.9, IQR 17.7-119.9 vs 45.1, IQR 17.8-167.1; <italic>P</italic>=.07; GPT-4 paired median 39.9, IQR 17.7-119.9 vs 30.7, IQR 11.5-91.7; <italic>P</italic>=.17).</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this preliminary evaluation of 392 published case reports, both the case report&#x2013;defined final diagnosis and the correct LLM-generated diagnosis verified by physicians had lower conditional perplexity scores than incorrect generated diagnoses. These findings indicate that physician-verified correctness was associated with greater conditional predictability under the independent scoring model; however, the present design does not establish case-specific clinical compatibility.</p><p>The within-case analyses also clarify the potential role of conditional perplexity in ranking differential diagnoses. Although the physician-verified correct diagnosis had lower conditional perplexity than the mean incorrect diagnosis in approximately 88%&#x2010;91% of evaluable lists, the mean is a permissive comparator because incorrect-diagnosis perplexity distributions were highly right-skewed. Against the more stringent lowest-perplexity incorrect candidate, the correct diagnosis had the lower score in only approximately one-quarter of evaluable lists across the 3 source models. Thus, conditional perplexity showed a distribution-level association with diagnostic correctness but did not reliably identify the correct diagnosis as the top-ranked candidate within an individual list. The present findings therefore do not support conditional perplexity as a stand-alone reranking or post-generation filtering rule; rather, it may warrant further investigation as one component of broader output-level auditing or multimodal rescoring approaches.</p><p>The comparison across source models is also informative. GPT-4 had the highest numerical rate of including the correct diagnosis in the differential diagnosis list and showed the lowest median conditional perplexity score for correct diagnoses and the highest candidate-level AUC among the 3 models. One possible interpretation is that GPT-4&#x2013;generated diagnosis strings were, on average, more closely aligned with the information captured in the case descriptions. Another possibility is that the diagnosis string used by GPT-4 was more naturally scored by the independent evaluator. Either way, the findings indicate that conditional perplexity score reflects properties of both narrative-to-diagnosis compatibility and language-model representation.</p><p>In paired analyses, the conditional perplexity scores for correct diagnoses generated by LLaMA2 and GPT-4 did not differ significantly from those of the case report&#x2013;defined final diagnoses, whereas Bard showed significantly higher scores. This finding indicates similar conditional predictability under the scoring model for the former 2 source models but should not be interpreted as evidence of equivalent case-specific clinical compatibility.</p><p>Additionally, the discrimination achieved by conditional perplexity score was moderate rather than near perfect. The ROC curves showed substantial overlap between correct and incorrect diagnoses, and some incorrect diagnoses still received low scores. This behavior is expected in real-world differential diagnosis, where several incorrect differential diagnoses may nonetheless be partially compatible with the presenting syndrome [<xref ref-type="bibr" rid="ref35">35</xref>]. Accordingly, the conditional perplexity score should be interpreted as a supportive output-level auditing signal rather than as a binary evaluation. In the context of XAI and uncertainty-aware decision support, such an externally computable score may add transparency without being mistaken for a complete explanation of clinical reasoning [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>These results align with the broader literature on LLMs in medicine, which shows strong promise in benchmark-style tasks while also highlighting the need for better evaluation of grounding, interpretability, and uncertainty [<xref ref-type="bibr" rid="ref48">48</xref>]. Recent work on hallucination and LLM reliability has distinguished data-driven or familiarity-based signals from reasoning-driven components [<xref ref-type="bibr" rid="ref49">49</xref>], a distinction that is directly relevant here because conditional perplexity is expected to track model familiarity and lexical predictability as well as conditional predictability. More broadly, output-based evaluations across multiple LLMs have demonstrated that externally defined quantitative measures can identify systematic differences in model-generated outputs that are not captured by conventional performance benchmarks [<xref ref-type="bibr" rid="ref50">50</xref>]. Although that work evaluated value-priority profiles rather than diagnostic reasoning, it illustrates the broader utility of output-level auditing across models. The present approach similarly applies an external quantitative measure to outputs from multiple source LLMs, but conditional perplexity remains task-specific and scorer-dependent. Similarly, calls for retrieval-grounded and verifiability-oriented evaluation of clinical LLM outputs emphasize that model-native scores should be interpreted alongside external evidence and subgroup or robustness audits rather than as stand-alone proof of validity [<xref ref-type="bibr" rid="ref51">51</xref>].</p></sec><sec id="s4-2"><title>Limitations</title><p>Several limitations should be acknowledged. First, this was a retrospective study using published case reports from a single journal and a single publication year, which may overrepresent uncommon, educational, or diagnostically polished cases. Published case reports are written after the diagnosis is known and may contain lexical cues that foreshadow the final diagnosis. As a result, the findings may not generalize to routine clinical reasoning, unfiltered electronic health record data, or prospective settings in which diagnostic information is incomplete, noisy, abbreviated, or evolving [<xref ref-type="bibr" rid="ref52">52</xref>].</p><p>Second, the analysis depended on diagnosis strings and prior physician verification in the previous dataset. Synonym handling, disease granularity, and adjudication choices may influence whether a generated diagnosis is counted as correct.</p><p>Third, the scorer model was Qwen2.5&#x2010;1.5B, a general-purpose LLM rather than a clinically specialized probability model, and conditional perplexity score may be influenced by diagnostic wording length, lexical frequency, or stylistic alignment in addition to clinical content [<xref ref-type="bibr" rid="ref53">53</xref>]. Although post hoc analyses adjusted for diagnosis-string length, lexical overlap, exact phrase occurrence, and a candidate-corpus token-frequency proxy, these analyses could not measure the true token frequency in the scorer model&#x2019;s pretraining distribution. Because only one independent scorer model was used, the robustness of the signal across different scorer architectures, larger models, and biomedical-domain scorers remains unknown.</p><p>Fourth, scorer-model contamination cannot be excluded. Qwen2.5&#x2010;1.5B may have been pretrained on some published case reports or related web content, and low perplexity for the correct diagnosis could partly reflect memorization or prior exposure rather than generalizable case-specific reasoning. Future studies should include contamination checks, held-out cases published after the scorer model training cutoff, or deidentified prospective clinical notes unavailable during pretraining.</p><p>Fifth, false-positive diagnoses in a differential diagnosis list are not necessarily nonsensical. Many may be partially compatible with alternatives that a clinician would reasonably entertain. In actual clinical practice, clinicians often intentionally include low-probability diagnoses when they represent life-threatening or require urgent exclusion [<xref ref-type="bibr" rid="ref54">54</xref>]. This likely lowers the apparent discrimination of the metric.</p><p>Sixth, the study evaluated text compatibility rather than clinical utility, so the findings should not be directly interpreted as demonstrating improved patient outcomes or safe autonomous diagnosis [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]. Case-level cluster-bootstrap sensitivity analyses were added to address nesting by case, but candidate-level pooled statistics and multiple pairwise tests remain exploratory and hypothesis-generating.</p><p>Finally, the rapid evolution of LLMs should be considered when interpreting these findings. Because the models examined in this study have advanced substantially, including Bard&#x2019;s transition to Gemini [<xref ref-type="bibr" rid="ref57">57</xref>], the release of LLaMA4 [<xref ref-type="bibr" rid="ref58">58</xref>], and the introduction of GPT-5 [<xref ref-type="bibr" rid="ref59">59</xref>], the present results may not directly extend to newer generations of LLMs.</p></sec><sec id="s4-3"><title>Future Directions</title><p>Despite these limitations, the findings support the conditional perplexity score as a promising quantitative measure for benchmarking diagnostic compatibility in case-report datasets. It adds information beyond binary inclusion, aligns directionally with physician-verified diagnostic status, and can be applied uniformly across differential diagnosis lists from multiple LLMs. Future studies should build on the post hoc sensitivity analyses by using scorer-model token frequencies or unconditional diagnosis-string likelihoods, masking diagnosis-revealing terms, matching correct and incorrect strings on length and frequency, and scoring diagnoses against mismatched case descriptions as a negative control. Additional validation should examine medically specialized scorer models, ontology-normalized diagnosis strings, combined rescoring strategies, and unfiltered real-world clinical documentation such as electronic health record data, where narratives are abbreviated, incomplete, and less retrospectively polished [<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref61">61</xref>]. It will also be important to determine whether this approach can be extended beyond free-text diagnostic strings to structured diagnostic representations, including International Classification of Diseases codes [<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref63">63</xref>], the systematized nomenclature of medicine clinical terms [<xref ref-type="bibr" rid="ref64">64</xref>], and, in Japan, Diagnosis Procedure Combination-related coding systems [<xref ref-type="bibr" rid="ref65">65</xref>], to improve standardization and reduce ambiguity in diagnosis matching. Because diagnosis is not determined by a single score alone, future research should also investigate how conditional perplexity score can be integrated with CDSS benchmarks and established diagnostic evaluation frameworks, so that compatibility scoring is interpreted as one component of a broader diagnostic assessment rather than as a standalone indicator.</p></sec><sec id="s4-4"><title>Conclusions</title><p>Conditional perplexity score derived from an independent LLM provided a moderate signal separating case report&#x2013;defined and physician-verified correct generated diagnoses from incorrect generated diagnoses in a large set of curated published case reports. The signal was consistent across 3 source LLMs and strongest for GPT-4&#x2013;generated differential diagnoses, but the magnitude of discrimination was moderate. These findings support conditional perplexity score as a promising quantitative adjunct for ranking, auditing, and studying LLM-generated differential diagnoses, while highlighting the need for prospective validation in broader and less curated clinical datasets.</p></sec></sec></body><back><ack><p>This study was made possible using the resources from the Department of Diagnostic and Generalist Medicine, Dokkyo Medical University.</p><p>ChatGPT and Gemini were used to suggest language improvements in the manuscript. These tools were not used to generate the study data, conduct statistical analyses, interpret the results, or make scientific conclusions. All AI-assisted language suggestions were reviewed, edited, and verified by the authors, who take full responsibility for the final content of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was supported by JSPS KAKENHI Grant Number JP24K20178 and JP26K13028.</p></sec><sec><title>Data Availability</title><p>The source case reports are publicly available through the <italic>American Journal of Case Reports</italic>. The analysis code and computational environment are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1</xref><xref ref-type="supplementary-material" rid="app2"/>-<xref ref-type="supplementary-material" rid="app3">3</xref>.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: TH</p><p>Data curation: TH</p><p>Formal analysis: TH</p><p>Funding acquisition: TH</p><p>Investigation: TH</p><p>Methodology: TH</p><p>Project administration: TH</p><p>Resources: TH</p><p>Writing &#x2013; original draft: TH</p><p>Writing &#x2013; review and editing: TS</p><p>All authors including YH, RK, and TS have read and agreed to the published version of the manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb2">CDSS</term><def><p>clinical decision support system</p></def></def-item><def-item><term id="abb3">CPU</term><def><p>central processing unit</p></def></def-item><def-item><term id="abb4">GPU</term><def><p>graphics processing unit</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">NLL</term><def><p>negative log-likelihood</p></def></def-item><def-item><term id="abb7">R-IDEA</term><def><p>revised-IDEA</p></def></def-item><def-item><term id="abb8">ROC</term><def><p>receiver operating characteristic</p></def></def-item><def-item><term id="abb9">ROC</term><def><p>receiver operating characteristic</p></def></def-item><def-item><term id="abb10">XAI</term><def><p>explainable artificial intelligence</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="book"><person-group person-group-type="author"><collab>Committee on Diagnostic Error in Health Care; Board on Health Care Services; Institute of Medicine; The National Academies of Sciences, Engineering, and Medicine</collab></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Balogh</surname><given-names>EP</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>BT</given-names> </name><name name-style="western"><surname>Ball</surname><given-names>JR</given-names> </name></person-group><source>Improving Diagnosis in Health Care</source><year>2015</year><publisher-name>National Academies Press</publisher-name><pub-id pub-id-type="doi">10.17226/21794</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Newman-Toker</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Nassery</surname><given-names>N</given-names> </name><name name-style="western"><surname>Schaffer</surname><given-names>AC</given-names> </name><etal/></person-group><article-title>Burden of serious harms from diagnostic error in the USA</article-title><source>BMJ Qual Saf</source><year>2024</year><month>01</month><day>19</day><volume>33</volume><issue>2</issue><fpage>109</fpage><lpage>120</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2021-014130</pub-id><pub-id pub-id-type="medline">37460118</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Graber</surname><given-names>M</given-names> </name></person-group><article-title>Diagnostic errors in medicine: a case of neglect</article-title><source>Jt Comm J Qual Patient Saf</source><year>2005</year><month>02</month><volume>31</volume><issue>2</issue><fpage>106</fpage><lpage>113</lpage><pub-id pub-id-type="doi">10.1016/s1553-7250(05)31015-4</pub-id><pub-id pub-id-type="medline">15791770</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Graber</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Franklin</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gordon</surname><given-names>R</given-names> </name></person-group><article-title>Diagnostic error in internal medicine</article-title><source>Arch Intern Med</source><year>2005</year><month>07</month><day>11</day><volume>165</volume><issue>13</issue><fpage>1493</fpage><lpage>1499</lpage><pub-id pub-id-type="doi">10.1001/archinte.165.13.1493</pub-id><pub-id pub-id-type="medline">16009864</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schiff</surname><given-names>GD</given-names> </name><name name-style="western"><surname>Hasan</surname><given-names>O</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Diagnostic error in medicine: analysis of 583 physician-reported errors</article-title><source>Arch Intern Med</source><year>2009</year><month>11</month><day>9</day><volume>169</volume><issue>20</issue><fpage>1881</fpage><lpage>1887</lpage><pub-id pub-id-type="doi">10.1001/archinternmed.2009.333</pub-id><pub-id pub-id-type="medline">19901140</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ely</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Kaldjian</surname><given-names>LC</given-names> </name><name name-style="western"><surname>D&#x2019;Alessandro</surname><given-names>DM</given-names> </name></person-group><article-title>Diagnostic errors in primary care: lessons learned</article-title><source>J Am Board Fam Med</source><year>2012</year><volume>25</volume><issue>1</issue><fpage>87</fpage><lpage>97</lpage><pub-id pub-id-type="doi">10.3122/jabfm.2012.01.110174</pub-id><pub-id pub-id-type="medline">22218629</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Giardina</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Meyer</surname><given-names>AND</given-names> </name><name name-style="western"><surname>Forjuoh</surname><given-names>SN</given-names> </name><name name-style="western"><surname>Reis</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>EJ</given-names> </name></person-group><article-title>Types and origins of diagnostic errors in primary care settings</article-title><source>JAMA Intern Med</source><year>2013</year><month>03</month><day>25</day><volume>173</volume><issue>6</issue><fpage>418</fpage><lpage>425</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2013.2777</pub-id><pub-id pub-id-type="medline">23440149</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Norman</surname><given-names>GR</given-names> </name><name name-style="western"><surname>Monteiro</surname><given-names>SD</given-names> </name><name name-style="western"><surname>Sherbino</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ilgen</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Schmidt</surname><given-names>HG</given-names> </name><name name-style="western"><surname>Mamede</surname><given-names>S</given-names> </name></person-group><article-title>The causes of errors in clinical reasoning: cognitive biases, knowledge deficits, and dual process thinking</article-title><source>Acad Med</source><year>2017</year><month>01</month><volume>92</volume><issue>1</issue><fpage>23</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.1097/ACM.0000000000001421</pub-id><pub-id pub-id-type="medline">27782919</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Berwick</surname><given-names>DM</given-names> </name></person-group><article-title>Diagnostic excellence through the lens of patient-centeredness</article-title><source>JAMA</source><year>2021</year><month>12</month><day>7</day><volume>326</volume><issue>21</issue><fpage>2127</fpage><lpage>2128</lpage><pub-id pub-id-type="doi">10.1001/jama.2021.19513</pub-id><pub-id pub-id-type="medline">34792525</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Fineberg</surname><given-names>HV</given-names> </name><name name-style="western"><surname>Cosby</surname><given-names>K</given-names> </name></person-group><article-title>Diagnostic Excellence</article-title><source>JAMA</source><year>2021</year><month>11</month><day>16</day><volume>326</volume><issue>19</issue><fpage>1905</fpage><lpage>1906</lpage><pub-id pub-id-type="doi">10.1001/jama.2021.19493</pub-id><pub-id pub-id-type="medline">34709367</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Morgan</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Srinivasan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bradford</surname><given-names>A</given-names> </name><name name-style="western"><surname>McDonald</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Kutty</surname><given-names>PK</given-names> </name></person-group><article-title>CDC&#x2019;s Core Elements to promote diagnostic excellence</article-title><source>Diagnosis (Berl)</source><year>2025</year><month>05</month><day>1</day><volume>12</volume><issue>2</issue><fpage>197</fpage><lpage>200</lpage><pub-id pub-id-type="doi">10.1515/dx-2024-0163</pub-id><pub-id pub-id-type="medline">39602334</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sutton</surname><given-names>RT</given-names> </name><name name-style="western"><surname>Pincock</surname><given-names>D</given-names> </name><name name-style="western"><surname>Baumgart</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Sadowski</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Fedorak</surname><given-names>RN</given-names> </name><name name-style="western"><surname>Kroeker</surname><given-names>KI</given-names> </name></person-group><article-title>An overview of clinical decision support systems: benefits, risks, and strategies for success</article-title><source>NPJ Digit Med</source><year>2020</year><volume>3</volume><issue>1</issue><fpage>17</fpage><pub-id pub-id-type="doi">10.1038/s41746-020-0221-y</pub-id><pub-id pub-id-type="medline">32047862</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Baalen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Boon</surname><given-names>M</given-names> </name><name name-style="western"><surname>Verhoef</surname><given-names>P</given-names> </name></person-group><article-title>From clinical decision support to clinical reasoning support systems</article-title><source>J Eval Clin Pract</source><year>2021</year><month>06</month><volume>27</volume><issue>3</issue><fpage>520</fpage><lpage>528</lpage><pub-id pub-id-type="doi">10.1111/jep.13541</pub-id><pub-id pub-id-type="medline">33554432</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Shortliffe</surname><given-names>EH</given-names> </name></person-group><source>Computer-Based Medical Consultations: MYCIN</source><year>1976</year><publisher-name>Elsevier</publisher-name><pub-id pub-id-type="other">9780444001795</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haigh</surname><given-names>T</given-names> </name></person-group><article-title>Between the Booms: AI in Winter</article-title><source>Commun ACM</source><year>2024</year><month>11</month><volume>67</volume><issue>11</issue><fpage>18</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.1145/3688379</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Sarikaya</surname><given-names>F</given-names> </name></person-group><article-title>The cycles of AI winters: a historical analysis and modern perspective</article-title><source>Zenodo</source><year>2024</year><access-date>2026-08-31</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/14015032">https://zenodo.org/records/14015032</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Knevel</surname><given-names>R</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>KP</given-names> </name></person-group><article-title>From real-world electronic health record data to real-world results using artificial intelligence</article-title><source>Ann Rheum Dis</source><year>2023</year><month>03</month><volume>82</volume><issue>3</issue><fpage>306</fpage><lpage>311</lpage><pub-id pub-id-type="doi">10.1136/ard-2022-222626</pub-id><pub-id pub-id-type="medline">36150748</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cairns</surname><given-names>BJ</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>T</given-names> </name></person-group><article-title>Generating synthetic mixed-type longitudinal electronic health records for artificial intelligent applications</article-title><source>NPJ Digit Med</source><year>2023</year><month>05</month><day>27</day><volume>6</volume><issue>1</issue><fpage>98</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00834-7</pub-id><pub-id pub-id-type="medline">37244963</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naveed</surname><given-names>H</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>AU</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A comprehensive overview of large language models</article-title><source>ACM Trans Intell Syst Technol</source><year>2025</year><month>10</month><day>31</day><volume>16</volume><issue>5</issue><fpage>1</fpage><lpage>72</lpage><pub-id pub-id-type="doi">10.1145/3744746</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Akkisetty</surname><given-names>PK</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>AMR</surname><given-names>PRC</given-names> </name><name name-style="western"><surname>Colby</surname><given-names>R</given-names> </name><name name-style="western"><surname>Nagasubramanian</surname><given-names>G</given-names> </name><name name-style="western"><surname>Ranganath</surname><given-names>S</given-names> </name></person-group><article-title>An overview of AI platforms, frameworks, libraries, and processors</article-title><source>Model Optimization Methods for Efficient and Edge AI: Federated Learning Architectures, Frameworks and Applications</source><year>2024</year><publisher-name>Wiley</publisher-name><fpage>43</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.1002/9781394219230</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kafkas</surname><given-names>&#x015E;</given-names> </name><name name-style="western"><surname>Abdelhakim</surname><given-names>M</given-names> </name><name name-style="western"><surname>Althagafi</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The application of large language models to the phenotype-based prioritization of causative genes in rare disease patients</article-title><source>Sci Rep</source><year>2025</year><month>04</month><day>29</day><volume>15</volume><issue>1</issue><fpage>15093</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-99539-y</pub-id><pub-id pub-id-type="medline">40301638</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lucas</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Upperman</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>JR</given-names> </name></person-group><article-title>A systematic review of large language models and their implications in medical education</article-title><source>Med Educ</source><year>2024</year><month>11</month><volume>58</volume><issue>11</issue><fpage>1276</fpage><lpage>1285</lpage><pub-id pub-id-type="doi">10.1111/medu.15402</pub-id><pub-id pub-id-type="medline">38639098</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cooper</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rodman</surname><given-names>A</given-names> </name></person-group><article-title>AI and medical education&#x2014;a 21st-century Pandora&#x2019;s box</article-title><source>N Engl J Med</source><year>2023</year><month>08</month><day>3</day><volume>389</volume><issue>5</issue><fpage>385</fpage><lpage>387</lpage><pub-id pub-id-type="doi">10.1056/NEJMp2304993</pub-id><pub-id pub-id-type="medline">37522417</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="editor"><name name-style="western"><surname>Gilpin</surname><given-names>LH</given-names> </name><name name-style="western"><surname>Bau</surname><given-names>D</given-names></name><name name-style="western"><surname>Yuan</surname><given-names>BZ</given-names></name><name name-style="western"><surname>Bajwa</surname><given-names>A</given-names> </name><name name-style="western"><surname>Specter</surname><given-names>M</given-names></name><name name-style="western"><surname>Kagal</surname><given-names>L</given-names></name></person-group><article-title>Explaining explanations: an overview of interpretability of machine learning</article-title><year>2018</year><conf-name>2018 IEEE 5th International Conference on Data Science and Advanced Analytics (DSAA)</conf-name><conf-date>Oct 1-3, 2018</conf-date><pub-id pub-id-type="doi">10.48550/arXiv.1806.00069</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sandmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Riepenhausen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Plagwitz</surname><given-names>L</given-names> </name><name name-style="western"><surname>Varghese</surname><given-names>J</given-names> </name></person-group><article-title>Systematic analysis of ChatGPT, Google search and Llama 2 for clinical decision support tasks</article-title><source>Nat Commun</source><year>2024</year><month>03</month><day>6</day><volume>15</volume><issue>1</issue><fpage>2050</fpage><pub-id pub-id-type="doi">10.1038/s41467-024-46411-8</pub-id><pub-id pub-id-type="medline">38448475</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Savage</surname><given-names>T</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Large language model uncertainty proxies: discrimination and calibration for medical diagnosis and treatment</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>01</month><day>1</day><volume>32</volume><issue>1</issue><fpage>139</fpage><lpage>149</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae254</pub-id><pub-id pub-id-type="medline">39396184</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bentegeac</surname><given-names>R</given-names> </name><name name-style="western"><surname>Le Guellec</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kuchcinski</surname><given-names>G</given-names> </name><name name-style="western"><surname>Amouyel</surname><given-names>P</given-names> </name><name name-style="western"><surname>Hamroun</surname><given-names>A</given-names> </name></person-group><article-title>Token probabilities to mitigate large language models overconfidence in answering medical questions: quantitative study</article-title><source>J Med Internet Res</source><year>2025</year><month>08</month><day>29</day><volume>27</volume><fpage>e64348</fpage><pub-id pub-id-type="doi">10.2196/64348</pub-id><pub-id pub-id-type="medline">40882190</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>W</given-names> </name><etal/></person-group><article-title>A survey on hallucination in large language models: principles, taxonomy, challenges, and open questions</article-title><source>ACM Trans Inf Syst</source><year>2025</year><volume>43</volume><issue>2</issue><fpage>1</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.1145/3703155</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Berg</surname><given-names>HT</given-names> </name><name name-style="western"><surname>van Bakel</surname><given-names>B</given-names> </name><name name-style="western"><surname>van de Wouw</surname><given-names>L</given-names> </name><etal/></person-group><article-title>ChatGPT and generating a differential diagnosis early in an emergency department presentation</article-title><source>Ann Emerg Med</source><year>2024</year><month>01</month><volume>83</volume><issue>1</issue><fpage>83</fpage><lpage>86</lpage><pub-id pub-id-type="doi">10.1016/j.annemergmed.2023.08.003</pub-id><pub-id pub-id-type="medline">37690022</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bridges</surname><given-names>JM</given-names> </name></person-group><article-title>Computerized diagnostic decision support systems&#x2014;a comparative performance study of Isabel Pro vs. ChatGPT4</article-title><source>Diagnosis (Berl)</source><year>2024</year><month>08</month><day>1</day><volume>11</volume><issue>3</issue><fpage>250</fpage><lpage>258</lpage><pub-id pub-id-type="doi">10.1515/dx-2024-0033</pub-id><pub-id pub-id-type="medline">38709491</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kanjee</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Crowe</surname><given-names>B</given-names> </name><name name-style="western"><surname>Rodman</surname><given-names>A</given-names> </name></person-group><article-title>Accuracy of a generative artificial intelligence model in a complex diagnostic challenge</article-title><source>JAMA</source><year>2023</year><month>07</month><day>3</day><volume>330</volume><issue>1</issue><fpage>78</fpage><lpage>80</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.8288</pub-id><pub-id pub-id-type="medline">37318797</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Takita</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kabata</surname><given-names>D</given-names> </name><name name-style="western"><surname>Walston</surname><given-names>SL</given-names> </name><etal/></person-group><article-title>A systematic review and meta-analysis of diagnostic performance comparison between generative AI and physicians</article-title><source>NPJ Digit Med</source><year>2025</year><month>03</month><day>22</day><volume>8</volume><issue>1</issue><fpage>175</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01543-z</pub-id><pub-id pub-id-type="medline">40121370</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schaye</surname><given-names>V</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>L</given-names> </name><name name-style="western"><surname>Kudlowitz</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Development of a clinical reasoning documentation assessment tool for resident and fellow admission notes: a shared mental model for feedback</article-title><source>J Gen Intern Med</source><year>2022</year><month>02</month><volume>37</volume><issue>3</issue><fpage>507</fpage><lpage>512</lpage><pub-id pub-id-type="doi">10.1007/s11606-021-06805-6</pub-id><pub-id pub-id-type="medline">33945113</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Richens</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Johri</surname><given-names>S</given-names> </name></person-group><article-title>Improving the accuracy of medical diagnosis with causal machine learning</article-title><source>Nat Commun</source><year>2020</year><month>08</month><day>11</day><volume>11</volume><issue>1</issue><fpage>3923</fpage><pub-id pub-id-type="doi">10.1038/s41467-020-17419-7</pub-id><pub-id pub-id-type="medline">32782264</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Angelov</surname><given-names>PP</given-names> </name><name name-style="western"><surname>Soares</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Arnold</surname><given-names>NI</given-names> </name><name name-style="western"><surname>Atkinson</surname><given-names>PM</given-names> </name></person-group><article-title>Explainable artificial intelligence: an analytical review</article-title><source>WIREs Data Min Knowl</source><year>2021</year><month>09</month><volume>11</volume><issue>5</issue><fpage>e1424</fpage><pub-id pub-id-type="doi">10.1002/widm.1424</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nazar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alam</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Yafi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Su&#x2019;ud</surname><given-names>MM</given-names> </name></person-group><article-title>A systematic review of human&#x2013;computer interaction and explainable artificial intelligence in healthcare with artificial intelligence techniques</article-title><source>IEEE Access</source><year>2021</year><volume>9</volume><fpage>153316</fpage><lpage>153348</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2021.3127881</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Begoli</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bhattacharya</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kusnezov</surname><given-names>D</given-names> </name></person-group><article-title>The need for uncertainty quantification in machine-assisted medical decision making</article-title><source>Nat Mach Intell</source><year>2019</year><volume>1</volume><fpage>20</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.1038/s42256-018-0004-1</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jelinek</surname><given-names>F</given-names> </name><name name-style="western"><surname>Mercer</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Bahl</surname><given-names>LR</given-names> </name><name name-style="western"><surname>Baker</surname><given-names>JK</given-names> </name></person-group><article-title>Perplexity&#x2014;a measure of the difficulty of speech recognition tasks</article-title><source>J Acoust Soc Am</source><year>1977</year><month>12</month><day>1</day><volume>62</volume><issue>S1</issue><fpage>S63</fpage><lpage>S63</lpage><pub-id pub-id-type="doi">10.1121/1.2016299</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Klakow</surname><given-names>D</given-names> </name><name name-style="western"><surname>Peters</surname><given-names>J</given-names> </name></person-group><article-title>Testing the correlation of word error rate and perplexity</article-title><source>Speech Commun</source><year>2002</year><month>09</month><volume>38</volume><issue>1-2</issue><fpage>19</fpage><lpage>28</lpage><pub-id pub-id-type="doi">10.1016/S0167-6393(01)00041-3</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Fang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Jegelka</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>J</given-names> </name><etal/></person-group><article-title>What is wrong with perplexity for long-context language modeling?</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 31, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2410.23771</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>P</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>K</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Harnessing multiple large language models: a survey on LLM ensemble</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 25, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2502.18036</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hirosawa</surname><given-names>T</given-names> </name><name name-style="western"><surname>Harada</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Mizuta</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sakamoto</surname><given-names>T</given-names> </name><name name-style="western"><surname>Tokumasu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Shimizu</surname><given-names>T</given-names> </name></person-group><article-title>Diagnostic performance of generative artificial intelligences for a series of complex case reports</article-title><source>Digit Health</source><year>2024</year><volume>10</volume><fpage>20552076241265215</fpage><pub-id pub-id-type="doi">10.1177/20552076241265215</pub-id><pub-id pub-id-type="medline">39229463</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Hui</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>B</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen2. 5 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 19, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.15115</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Jeon</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ko</surname><given-names>G</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ha</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ro</surname><given-names>WW</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Kim</surname><given-names>S</given-names></name><name name-style="western"><surname>Deka</surname><given-names>GC</given-names></name></person-group><article-title>Deep learning with GPUs</article-title><source>Advances in Computers</source><year>2021</year><publisher-name>Elsevier</publisher-name><fpage>167</fpage><lpage>215</lpage><pub-id pub-id-type="doi">10.1016/bs.adcom.2020.11.003</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bimbot</surname><given-names>F</given-names> </name><name name-style="western"><surname>El-B&#x00E8;ze</surname><given-names>M</given-names> </name><name name-style="western"><surname>Igounet</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jardino</surname><given-names>M</given-names> </name><name name-style="western"><surname>Smaili</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zitouni</surname><given-names>I</given-names> </name></person-group><article-title>An alternative scheme for perplexity estimation and its assessment for the evaluation of language models</article-title><source>Comput Speech Lang</source><year>2001</year><month>01</month><volume>15</volume><issue>1</issue><fpage>1</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1006/csla.2000.0150</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steyerberg</surname><given-names>EW</given-names> </name><name name-style="western"><surname>Harrell</surname><given-names>FE</given-names> </name><name name-style="western"><surname>Borsboom</surname><given-names>GJ</given-names> </name><name name-style="western"><surname>Eijkemans</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Vergouwe</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Habbema</surname><given-names>JD</given-names> </name></person-group><article-title>Internal validation of predictive models: efficiency of some procedures for logistic regression analysis</article-title><source>J Clin Epidemiol</source><year>2001</year><month>08</month><volume>54</volume><issue>8</issue><fpage>774</fpage><lpage>781</lpage><pub-id pub-id-type="doi">10.1016/s0895-4356(01)00341-9</pub-id><pub-id pub-id-type="medline">11470385</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hertzmann</surname><given-names>A</given-names> </name><name name-style="western"><surname>Russakovsky</surname><given-names>O</given-names> </name></person-group><article-title>Benchmark suites instead of leaderboards for evaluating AI fairness</article-title><source>Patterns (N Y)</source><year>2024</year><month>11</month><day>8</day><volume>5</volume><issue>11</issue><fpage>101080</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2024.101080</pub-id><pub-id pub-id-type="medline">39568473</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zeng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>HalluGuard: demystifying data-driven and reasoning-driven hallucinations in LLMs</article-title><access-date>2026-08-26</access-date><conf-name>International Conference on Learning Representations (ICLR) 2026</conf-name><conf-date>Apr 23-25, 2026</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=ZURs3YZclt">https://openreview.net/pdf?id=ZURs3YZclt</ext-link></comment></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lau</surname><given-names>GR</given-names> </name><name name-style="western"><surname>Low</surname><given-names>WY</given-names> </name><name name-style="western"><surname>Koh</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Nah</surname><given-names>FFH</given-names> </name><name name-style="western"><surname>Hartanto</surname><given-names>A</given-names> </name></person-group><article-title>Evaluating AI alignment in LLMs: output analysis of value priorities across 75 models with human benchmarking</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 14, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2506.12617</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name></person-group><article-title>Toward retrieval-grounded evaluation for conversational large language model-based risk assessment</article-title><source>JMIR AI</source><year>2026</year><month>03</month><day>12</day><volume>5</volume><fpage>e90759</fpage><pub-id pub-id-type="doi">10.2196/90759</pub-id><pub-id pub-id-type="medline">41818631</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Beyond multiple-choice questions: rethinking evaluation frameworks for large language models for clinical medicine</article-title><source>Intell Med</source><year>2026</year><month>04</month><volume>6</volume><issue>2</issue><fpage>109</fpage><lpage>115</lpage><pub-id pub-id-type="doi">10.1016/j.imed.2026.01.001</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jacoby</surname><given-names>A</given-names> </name><name name-style="western"><surname>Goldberg</surname><given-names>Y</given-names> </name></person-group><article-title>Same task, more tokens: the impact of input length on the reasoning performance of large language models</article-title><source>Proc Annu Meet Assoc Comput Linguist</source><year>2024</year><fpage>15339</fpage><lpage>15353</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.acl-long.818</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hsia</surname><given-names>RY</given-names> </name><name name-style="western"><surname>Hale</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Tabas</surname><given-names>JA</given-names> </name></person-group><article-title>A national study of the prevalence of life-threatening diagnoses in patients with chest pain</article-title><source>JAMA Intern Med</source><year>2016</year><month>07</month><day>1</day><volume>176</volume><issue>7</issue><fpage>1029</fpage><lpage>1032</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2016.2498</pub-id><pub-id pub-id-type="medline">27295579</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Afroogh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Akbari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Malone</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kargar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alambeigi</surname><given-names>H</given-names> </name></person-group><article-title>Trust in AI: progress, challenges, and future directions</article-title><source>Humanit Soc Sci Commun</source><year>2024</year><volume>11</volume><issue>1</issue><fpage>1568</fpage><pub-id pub-id-type="doi">10.1057/s41599-024-04044-8</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Challen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Denny</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pitt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gompels</surname><given-names>L</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>T</given-names> </name><name name-style="western"><surname>Tsaneva-Atanasova</surname><given-names>K</given-names> </name></person-group><article-title>Artificial intelligence, bias and clinical safety</article-title><source>BMJ Qual Saf</source><year>2019</year><month>03</month><volume>28</volume><issue>3</issue><fpage>231</fpage><lpage>237</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2018-008370</pub-id><pub-id pub-id-type="medline">30636200</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>Gemini Team Google</collab><name name-style="western"><surname>Anil</surname><given-names>R</given-names> </name><name name-style="western"><surname>Borgeaud</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Alayrac</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Gemini: a family of highly capable multimodal models</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 19, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2312.11805</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="web"><source>Llama</source><access-date>2026-08-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.llama.com/">https://www.llama.com/</ext-link></comment></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="web"><article-title>Introducing GPT-5.2</article-title><source>OpenAI</source><year>2025</year><access-date>2025-12-11</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/index/introducing-gpt-5-2">https://openai.com/index/introducing-gpt-5-2</ext-link></comment></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Madan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lentzen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Brandt</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rueckert</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hofmann-Apitius</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fr&#x00F6;hlich</surname><given-names>H</given-names> </name></person-group><article-title>Transformer models in biomedicine</article-title><source>BMC Med Inform Decis Mak</source><year>2024</year><month>07</month><day>29</day><volume>24</volume><issue>1</issue><fpage>214</fpage><pub-id pub-id-type="doi">10.1186/s12911-024-02600-5</pub-id><pub-id pub-id-type="medline">39075407</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Saab</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Tanno</surname><given-names>R</given-names> </name><name name-style="western"><surname>Stutz</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wulczyn</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Capabilities of Gemini models in medicine</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 29, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2404.18416</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Gifford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Mapping ICD-10 and ICD-10-CM codes to phecodes: workflow development and initial evaluation</article-title><source>JMIR Med Inform</source><year>2019</year><month>11</month><day>29</day><volume>7</volume><issue>4</issue><fpage>e14325</fpage><pub-id pub-id-type="doi">10.2196/14325</pub-id><pub-id pub-id-type="medline">31553307</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kirby</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Speltz</surname><given-names>P</given-names> </name><name name-style="western"><surname>Rasmussen</surname><given-names>LV</given-names> </name><etal/></person-group><article-title>PheKB: a catalog and workflow for creating electronic phenotype algorithms for transportability</article-title><source>J Am Med Inform Assoc</source><year>2016</year><month>11</month><volume>23</volume><issue>6</issue><fpage>1046</fpage><lpage>1052</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocv202</pub-id><pub-id pub-id-type="medline">27026615</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>D</given-names> </name><name name-style="western"><surname>de Keizer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lau</surname><given-names>F</given-names> </name><name name-style="western"><surname>Cornet</surname><given-names>R</given-names> </name></person-group><article-title>Literature review of SNOMED CT use</article-title><source>J Am Med Inform Assoc</source><year>2014</year><month>02</month><volume>21</volume><issue>e1</issue><fpage>e11</fpage><lpage>e19</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2013-001636</pub-id><pub-id pub-id-type="medline">23828173</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hayashida</surname><given-names>K</given-names> </name><name name-style="western"><surname>Murakami</surname><given-names>G</given-names> </name><name name-style="western"><surname>Matsuda</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fushimi</surname><given-names>K</given-names> </name></person-group><article-title>History and profile of diagnosis procedure combination (DPC): development of a real data collection system for acute inpatient care in Japan</article-title><source>J Epidemiol</source><year>2021</year><month>01</month><day>5</day><volume>31</volume><issue>1</issue><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.2188/jea.JE20200288</pub-id><pub-id pub-id-type="medline">33012777</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Sample Python code to calculate conditional perplexity scores.</p><media xlink:href="formative_v10i1e98819_app1.docx" xlink:title="DOCX File, 35 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Sample Python code for statistical analysis and post hoc sensitivity analyses.</p><media xlink:href="formative_v10i1e98819_app2.docx" xlink:title="DOCX File, 49 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Supplemental sensitivity analyses addressing lexical overlap, adjusted models, case-level cluster-bootstrap area under the receiver operating characteristic curves, and within-case comparator analyses.</p><media xlink:href="formative_v10i1e98819_app3.docx" xlink:title="DOCX File, 43 KB"/></supplementary-material></app-group></back></article>