<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e104579</article-id><article-id pub-id-type="doi">10.2196/104579</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Real-Time Artificial Intelligence Diagnostic Copilot in Simulated Primary Care Consultations: Randomized Simulation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Cusacovich</surname><given-names>Ivan</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Pinilla</surname><given-names>Manuel</given-names></name><degrees>MEng, MBA, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Garcia Castro</surname><given-names>Ruben</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>Internal Medicine Department, Hospital Cl&#x00ED;nico Universitario de Valladolid</institution><addr-line>Valladolid</addr-line><country>Spain</country></aff><aff id="aff2"><institution>Medsys AI, SL</institution><addr-line>Madrid</addr-line><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff3"><institution>Department of Medical-Surgical Dermatology and Venereology, Hospital Universitario Fundaci&#x00F3;n Jim&#x00E9;nez D&#x00ED;az</institution><addr-line>Avenida de los Reyes Cat&#x00F3;licos, 2</addr-line><addr-line>Madrid</addr-line><addr-line>Madrid</addr-line><country>Spain</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>MacNeill</surname><given-names>Luke</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Tsai</surname><given-names>Meng-Hsun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Isleyici</surname><given-names>Taha Kaan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Ruben Garcia Castro, MD, PhD, Department of Medical-Surgical Dermatology and Venereology, Hospital Universitario Fundaci&#x00F3;n Jim&#x00E9;nez D&#x00ED;az, Avenida de los Reyes Cat&#x00F3;licos, 2, Madrid, Madrid, 28040, Spain, 34 915504800; <email>rubengarciacastro.mail@gmail.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>all authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>22</day><month>9</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e104579</elocation-id><history><date date-type="received"><day>13</day><month>06</month><year>2026</year></date><date date-type="rev-recd"><day>02</day><month>09</month><year>2026</year></date><date date-type="accepted"><day>03</day><month>09</month><year>2026</year></date></history><copyright-statement>&#x00A9; Ivan Cusacovich, Manuel Pinilla, Ruben Garcia Castro. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 22.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e104579"/><abstract><sec><title>Background</title><p>Diagnostic errors in primary care contribute substantially to avoidable morbidity. Real-time, voice-based artificial intelligence (AI) diagnostic assistants may support diagnostic reasoning during clinical encounters, but formative evidence that can be used to assess both diagnostic benefit and overreliance risk before clinical deployment is needed from interactive settings.</p></sec><sec><title>Objective</title><p>This study aimed to conduct a formative, high-difficulty simulation stress test of whether a real-time, voice-based AI diagnostic assistant was associated with improved physician diagnostic accuracy and measurable overreliance during simulated primary care consultations.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a case-level randomized, adjudicator-blinded simulation study in a web-based virtual primary care clinic. Thirteen board-certified family and community medicine physicians managed 260 simulated voice-based consultations. Cases were assigned to either an AI-assisted condition or an unassisted, resource-restricted simulation condition without any external diagnostic aids. No real patients were enrolled, and no clinical care was delivered or modified. The primary outcome was Top-3 diagnostic accuracy, defined as at least one correct diagnosis among up to 3 submitted diagnoses assessed by 3 blinded adjudicators and analyzed using a frequentist binomial generalized linear mixed model fitted by maximum likelihood with random intercepts for physician and case.</p></sec><sec sec-type="results"><title>Results</title><p>All 260 consultations were completed and analyzed. Adjusted Top-3 diagnostic accuracy was 62.3% (95% CI 49.1% to 75.7%) in the unassisted condition and 74.6% (95% CI 63.7% to 85.4%) in the AI-assisted condition (adjusted odds ratio [AOR] 2.68, 95% CI 1.24 to 6.12; &#x03C7;<sup>2</sup><sub>1</sub>=6.4, <italic>P</italic>=.01). This corresponds to an adjusted absolute difference of +12.3 percentage points (95% CI +2.7 to +22.6) and a simulation-context number needed to treat equivalent of 8.1 (95% CI 4.4 to 37.2). The post hoc Top-2 analysis was supportive (AOR 2.59, 95% CI 1.23 to 5.77; &#x03C7;<sup>2</sup><sub>1</sub>=6.3, <italic>P</italic>=.01). Top-1 point estimates favored AI assistance but were inconclusive: Adjusted accuracy was 47.8% in the unassisted condition versus 57% with AI assistance (AOR 1.83, 95% CI 0.93 to 3.80; &#x03C7;<sup>2</sup><sub>1</sub>=3.1, <italic>P</italic>=.08). Leave-one-physician-out and virtual-patient simulator-exclusion sensitivity analyses were consistent with the primary finding. In risk scenarios with an incorrect operative AI suggestion, adjusted concordance was 38.8% in the unassisted condition and 55.8% in the AI-assisted condition, but the between-condition comparison was not statistically significant (AOR 2.08, 95% CI 0.89 to 5.19; &#x03C7;<sup>2</sup><sub>1</sub>=2.9, <italic>P</italic>=.09). Mean consultation time increased by 10.7%, and physicians rated the system highly for usefulness and satisfaction.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this randomized, high-difficulty simulation study, real-time, voice-based AI assistance was associated with higher physician Top-3 diagnostic accuracy than an unassisted, resource-restricted comparator. Some sensitivity analyses were supportive, while others were directionally favorable but inconclusive, and the safety analysis suggested possible overreliance without statistically conclusive difference between conditions. These findings support further refinement and prospective evaluation in real clinical workflows before conclusions are drawn about routine clinical effectiveness or implementation.</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>clinical decision support</kwd><kwd>diagnostic accuracy</kwd><kwd>automation bias</kwd><kwd>formative evaluation</kwd><kwd>simulation study</kwd><kwd>physician performance</kwd><kwd>primary care</kwd><kwd>human-AI interaction</kwd><kwd>virtual patients</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Diagnostic accuracy in primary care is critical, yet missed, delayed, or incorrect diagnoses remain common in ambulatory care and contribute to avoidable morbidity, fragmented follow-up, and increased health care costs [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Diagnostic difficulty is particularly relevant in primary care, where generalist physicians often evaluate broad, undifferentiated symptom constellations with limited testing at the point of care [<xref ref-type="bibr" rid="ref3">3</xref>]. Conditions with subtle, nonspecific, or atypical presentations, such as pulmonary embolism or systemic autoimmune disease, are especially vulnerable to delayed or missed diagnosis [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. These challenges create a rationale for tools that support hypothesis generation, information gathering, and diagnostic verification during the consultation.</p><p>Large language models (LLMs) have shown strong diagnostic performance in medical question-answering and curated vignette benchmarks, and recent systems increasingly move from single-answer generation toward conversational or workflow-level diagnostic support [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. However, evidence that LLMs improve physician performance is mixed. In a randomized vignette-based study, GPT-4 access did not significantly improve physicians&#x2019; diagnostic reasoning compared with conventional resources, despite strong standalone model performance [<xref ref-type="bibr" rid="ref9">9</xref>]. A later randomized trial found a more modest improvement in patient care tasks when GPT-4 support was embedded in a sequential information environment [<xref ref-type="bibr" rid="ref10">10</xref>]. More recent evidence suggests that gains may depend on structured AI-literacy training or collaborative diagnostic workflows rather than model access alone [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>Safety evaluation is equally important. LLMs can generate incorrect, unsupported, or biased outputs, and prior studies have documented hallucination, demographic bias, and cognitive bias sensitivity in medical LLM outputs [<xref ref-type="bibr" rid="ref13">13</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. Human-artificial intelligence (AI) interaction studies also show that incorrect AI recommendations can degrade clinician performance even when correct recommendations improve it [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Interface design is therefore central: Explanations, confidence cues, and recommendation framing may influence trust, but explainability alone does not reliably prevent overreliance [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Most prior diagnostic support studies have relied on static text vignettes or structured tasks rather than real-time, voice-based consultations. This distinction matters because diagnostic reasoning during a consultation evolves dynamically as physicians ask questions, test hypotheses, and decide what information is relevant. Voice-based assistants and ambient clinical tools are increasingly entering clinical workflows, but much of the existing evidence focuses on documentation, operational efficiency, or constrained clinical tasks rather than diagnostic accuracy and overreliance during an unfolding consultation [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. LLM-powered virtual patients provide a controlled way to evaluate these interactions before exposure to real patients, preserving experimental control while approximating some features of clinical dialogue [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>].</p></sec><sec id="s1-2"><title>Study Objective</title><p>Against this background, we conducted a formative, case-level randomized, adjudicator-blinded simulation study to evaluate a real-time, voice-based diagnostic decision-support system during high-difficulty simulated primary care consultations. The study was designed as a diagnostic stress test rather than as an estimate of routine clinical effectiveness. By using interactive virtual-patient consultations, the study aimed to move beyond static vignette testing while preserving a controlled setting in which diagnostic benefit and overreliance risk could be evaluated before prospective clinical deployment.</p><p>The primary objective was to assess whether AI assistance to the physician was associated with higher Top-3 diagnostic accuracy than an unassisted simulation condition without external diagnostic aids. A secondary safety objective was to quantify whether physicians were more likely to submit diagnoses concordant with incorrect AI suggestions when those suggestions were visible. These objectives reflect 2 linked questions for formative evaluation: whether real-time AI support can improve diagnostic performance in a challenging simulated setting and whether visible incorrect suggestions create a measurable overreliance signal. Consultation duration, physician-reported usefulness, and overall satisfaction were assessed as formative workflow and acceptability measures; the remaining additional analyses were exploratory or sensitivity analyses.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>We conducted a formative, case-level randomized, adjudicator-blinded simulation study in a web-based virtual primary care clinic. The study was designed as a high-difficulty diagnostic simulation stress test: Physicians evaluated diagnostically challenging virtual patient cases, and outcomes measured physician diagnostic performance, workflow, acceptability, and overreliance signals rather than patient health outcomes. We hypothesized that real-time, voice-integrated AI assistance would be associated with higher diagnostic accuracy than an unassisted simulation condition without external diagnostic aids, while also increasing the risk that physicians would submit diagnoses concordant with incorrect AI suggestions when those suggestions were visible.</p><p>Physicians conducted voice-based consultations with a patient simulator powered by the GPT-4o Realtime API for low-latency voice interaction. Through case-specific prompts, the model was anchored to predefined demographics, symptoms, and permitted primary care physical examination and point-of-care information. The simulator was instructed to behave as a patient rather than as a clinical assistant, to answer only physician-initiated questions, and not to disclose diagnostic labels or unsolicited clinical reasoning. To approximate a constrained primary care visit, it processed requests for predefined primary care point-of-care tools and physical examinations and declined advanced investigations such as imaging or laboratory workups. Pre-study testing was used to check adherence to this constrained role, and a retrospective assessment of that adherence was done based on the transcripts of the conversations. The architecture separated the patient simulator from the AI assistant, ensuring that the AI assistant processed only the live physician-patient dialogue and not the underlying vignette text nor simulator prompts (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><p>For each incoming virtual-patient case, physicians were assigned either to real-time AI support or to the unassisted simulation condition. In the AI-assisted condition, diagnostic support was visible during the consultation; in the unassisted condition, physicians used the same virtual clinic but without AI support or external diagnostic aids. A panel of 3 independent adjudicators, all practicing family physicians, assessed the correctness of diagnoses. A diagnosis was considered correct based on predefined criteria that accepted exact matches, synonyms, or a broader clinical category only when the specific gold standard subtype could not be differentiated with the information available in a primary care setting. Final adjudication was determined by majority rule: When at least 2 of the 3 adjudicators provided the same judgment, their shared judgment was retained as the final classification; otherwise, the assessment was to be discarded.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Diagram of the patient simulator&#x2019;s architecture used for the study. LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e104579_fig01.png"/></fig></sec><sec id="s2-2"><title>Study Registration and Repository</title><p>Study registration was not applicable because this was a formative, randomized, simulation-based, physician performance study using virtual patient cases: No real patients were enrolled, no clinical care was delivered or modified, no identifiable patient health data were used, and no patient health outcomes or patient adverse events were measured. All outcomes were physician-level performance, time, usability, or automation-bias measures.</p><p>To support transparency and reproducibility, the full study protocol, reporting checklist, simulated clinical vignettes, dataset from all 260 virtual consultations, adjudicator evaluations, statistical analysis code and results, data dictionary, analytical transparency document, virtual patient performance audit, and sensitivity analysis outputs were made publicly available in a permanent repository [<xref ref-type="bibr" rid="ref27">27</xref>]. The analytical transparency document maps the public statistical code and console outputs to the corresponding manuscript results, enabling end-to-end third-party verification from the locked adjudicator-derived dataset.</p></sec><sec id="s2-3"><title>Ethical Considerations</title><p>Although the study involved physician participants, it did not involve real patients, the delivery or modification of clinical care, identifiable patient health data, or patient health outcomes and therefore did not fall within the categories requiring ethics committee approval under applicable local regulations. Following completion of the study, the study documentation was reviewed by the Research Ethics Committee of Hospital Universitario La Paz (Madrid, Spain), which issued a formal institutional determination confirming that ethics approval was not required (Reference 57/230428.9/26, dated August 11, 2026). The processing of participating physicians&#x2019; personal data complied with Regulation (EU) 2016/679 and Spanish Organic Law 3/2018 [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Data related to participating physicians were de-identified and stored in an access-restricted Google Drive folder accessible only to the study authors; no directly identifying participant information was included in the adjudication files or the public repository.</p><p>All participating physicians provided written informed consent before taking part in the study. Recruitment and consent materials described the evaluated tool, the simulated case format, the expected time commitment, random assignment to AI-assisted and unassisted simulation conditions, and the assessment of diagnostic accuracy. Participation was voluntary, and participants could discontinue participation. Participants received no financial compensation. Access to the evaluated system during the simulation was provided at no cost to participants.</p></sec><sec id="s2-4"><title>Simulated Patients and Physician Participants</title><p>We selected 40 diagnostically challenging adult clinical vignettes from a peer-reviewed library that was originally validated through supermajority agreement by a panel of 7 independent primary care physicians [<xref ref-type="bibr" rid="ref30">30</xref>]. This selection was intentional: The case set was used to stress-test diagnostic reasoning under conditions in which missed or delayed diagnoses are plausible, not to reproduce the prevalence or average difficulty of unselected primary care encounters. A physician and 2 LLM-based agents, which were built on a different architecture from the evaluated assistant, selected the cases to reduce content familiarity. The AI assistant was structurally blinded to the original vignette text and underlying simulator prompts, evaluating exclusively the de novo live dialogue dynamically steered by the physician.</p><p>Separately, physicians were recruited online through a URL shared with the Madrid regional health service (SERMAS) and social media. Eligibility required active clinical practice in Spain and specialty board certification; computer and internet literacy were implicit eligibility criteria because participation required use of a secure web application and voice interface. Recruitment materials described the evaluated tool, the simulated cases, the expected time commitment, and the AI-assisted and unassisted simulation conditions. A total of 13 board-certified family and community medicine physicians were accepted with no more than one per clinic to reduce cross-contamination. Each physician completed a 30-minute online orientation with the simulation interface before the study.</p></sec><sec id="s2-5"><title>Randomization and Masking</title><p>Computer-generated randomization occurred at the case level. For each physician, cases were randomly drawn from the 40-case pool and assigned with intended equal probability to either AI assistance or the unassisted simulation condition; no blocking or stratification by physician or case was used, so the achieved distribution could deviate from exact 1:1 balance by chance.</p><p>Physicians were necessarily unblinded to AI availability because diagnostic suggestions were visible in AI-assisted consultations and absent in unassisted consultations. The 3 independent adjudicators who assessed diagnostic correctness were blinded: They received only physicians&#x2019; diagnoses, AI assistant diagnostic suggestions (also generated in the background for the unassisted condition cases), and the vignette gold standard diagnoses, stripped of all metadata that could identify study arms or indicate whether the AI suggestion had been displayed to the physician. Data analysts worked from the adjudicators&#x2019; evaluations and were not blinded; however, they did not influence outcomes, as all statistical modeling was conducted on a locked dataset comprised strictly of the independent adjudicators&#x2019; blinded scorings. After blinded adjudication, analytical endpoints were generated through deterministic recoding of the locked adjudicator-derived labels; therefore, the analysts had no ability to alter the primary outcome grading. The statistical code, console outputs, and result-to-manuscript analytical transparency document are available in the public repository [<xref ref-type="bibr" rid="ref27">27</xref>].</p></sec><sec id="s2-6"><title>Procedures</title><p>In both conditions, physicians used the same voice interface to conduct simulated consultations. The study was fully web-based and remote: Participants accessed a secure web application from home and were instructed to use a headset in an uninterrupted, quiet environment. To define a controlled, resource-restricted contrast, physicians were instructed not to use nonstudy external diagnostic aids during either condition. The unassisted comparator was intended to isolate the effect of real-time AI assistance under the study conditions and was not intended to represent resource-enabled usual care. After each consultation, physicians submitted between 1 and 3 final diagnoses, which were securely stored. Consultation duration, time stamps, arm assignment, case identifiers, and submitted diagnoses were captured by the web application. After completing their assigned cases, physicians completed an online questionnaire assessing perceived usefulness and overall satisfaction on 5-point response scales. All metadata and arm identifiers were stripped before blinded adjudication. The AI assistant was used under a license provided by the manufacturer.</p><p>In the AI-assisted condition, physicians received real-time decision support from the Medsys AI-Clinical Assistant V1.0.0 (Medsys AI, SL), the evaluated system, without any functionality change, content update, or bug fix release after study commencement. The system displayed diagnostic suggestions and information-gathering prompts during the consultation while leaving final diagnostic responsibility with the physician (<xref ref-type="fig" rid="figure2">Figure 2</xref>). It transcribed the physician-patient conversation and routed it to an orchestration subsystem, which queried a research agent for clinical guideline insights and a diagnostic agent for hypothetical diagnoses. The interface passively displayed a dynamic differential diagnosis with brief rationales, contextually relevant history-taking prompts, and physical examination suggestions. Physicians could use, ignore, or override any displayed content. The system integrates off-the-shelf speech-to-text and generative LLMs, specifically GPT-4o and GPT-4o-mini, which were called concurrently at high frequencies to maintain real-time, low-latency performance. No fine-tuning of these models was performed. Instead, the system relied on retrieval-augmented generation (RAG) anchored to a curated corpus of over 1000 clinical practice guidelines accepted by Spanish medical associations.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Schematics of the high-level logic architecture and interfaces of the evaluated system. AI: artificial intelligence.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e104579_fig02.png"/></fig></sec><sec id="s2-7"><title>Outcomes</title><sec id="s2-7-1"><title>Primary Outcome: Top-3 Diagnostic Accuracy</title><p>The primary outcome was Top-3 diagnostic accuracy, defined as the proportion of simulated consultations in which the physician provided at least 1 correct diagnosis among their list of up to 3 submitted diagnoses compared with the predefined gold standard for the case. This definition assessed whether the correct diagnosis was present in the physician&#x2019;s final differential diagnosis rather than whether it was ranked first. Because the simulated consultations ended before access to advanced diagnostic testing, this endpoint was intended to capture whether the correct diagnosis was represented in the end-of-consultation differential that would guide the selection and interpretation of subsequent tests, rather than whether it had already been ranked first. Top-1 and Top-2 accuracy were examined post hoc as complementary measures of diagnostic prioritization.</p></sec><sec id="s2-7-2"><title>Secondary Safety Outcome: Concordance With Incorrect AI Suggestions</title><p>The secondary safety outcome was concordance with incorrect AI suggestions in risk scenarios, defined as consultations in which the AI&#x2019;s operative diagnostic suggestion was incorrect. Because the AI suggestions were displayed continuously throughout the consultation rather than sequentially, adoption was determined by clinical concordance between the first diagnosis submitted by the physician and the AI&#x2019;s suggestions classified as high likelihood. This secondary safety outcome compared concordance with incorrect AI suggestions across arms. In the AI-assisted condition, concordance with an incorrect displayed suggestion was interpreted as adoption of that suggestion; in the unassisted simulation condition, concordance with an incorrect background suggestion was interpreted as error concordance rather than true adoption. Because this was a simulated environment, no direct patient-related adverse events were monitored.</p></sec><sec id="s2-7-3"><title>Formative Workflow and Acceptability Measures</title><p>Formative workflow and acceptability measures included consultation duration, which was automatically recorded by the platform, and physician-reported usefulness and overall satisfaction, which were collected after study completion through an online questionnaire using 5-point response scales.</p></sec><sec id="s2-7-4"><title>Sensitivity and Exploratory Analyses</title><p>Sensitivity analyses included post hoc Top-1 and Top-2 diagnostic accuracy, leave-one-physician-out re-estimation, and exclusion of consultations flagged in the virtual patient performance audit. Top-1 diagnostic accuracy was defined as the proportion of consultations in which the physician&#x2019;s first submitted diagnosis matched the predefined gold standard. Top-2 diagnostic accuracy was defined as the proportion of consultations in which either the physician&#x2019;s first or second submitted diagnosis matched the predefined gold standard.</p><p>Exploratory analyses examined effect modification by case difficulty, physician ability, recent workload, and case position. We used 2 post hoc metrics: case difficulty (defined for each case as one minus the mean unassisted condition accuracy) and physician ability (defined as the physician&#x2019;s unassisted performance relative to the mean for their specific case mix). Moderation was assessed by testing for statistical interaction between these metrics and the study arm. We also evaluated potential fatigue or familiarization using the case&#x2019;s sequential position and recent workload and conducted an exploratory comparison of physician performance with standalone AI performance, using background AI outputs generated for all cases.</p></sec></sec><sec id="s2-8"><title>Virtual Patient Performance Review</title><p>Because simulator behavior was central to the validity of the study, we conducted a retrospective transcript-based audit of virtual patient adherence to its role, assessing diagnostic leakage, excessive helpfulness, fluency or conversational blocking, and adherence to the required physical examination response format [<xref ref-type="bibr" rid="ref27">27</xref>].</p></sec><sec id="s2-9"><title>Statistical Analysis</title><p>All hypothesis tests were 2-sided with <italic>&#x03B1;</italic>=.05. All randomized consultations were analyzed according to assigned condition. Case characteristics and case allocation by physician were summarized using counts and percentages. Physician sex was summarized using counts and percentages, whereas physician age and clinical experience were summarized using medians and IQRs. No imputation was required because all randomized consultations had complete outcome data.</p><sec id="s2-9-1"><title>Primary Outcome: Top-3 Diagnostic Accuracy</title><p>The primary endpoint (Top-3 diagnostic accuracy) was analyzed using a frequentist binomial generalized linear mixed model (GLMM) fitted by maximum likelihood with the Laplace approximation. The model included the randomized arm (AI-assisted or unassisted) as a fixed effect, with random intercepts for both physician and case to control for nonindependent observations. From this model, we derived the primary adjusted odds ratio (AOR). Throughout the GLMM analyses, fixed-effect contrasts and interaction terms of interest were evaluated using likelihood-ratio tests against the corresponding nested models. Each comparison differed by 1 fixed-effect parameter and therefore had 1 degree of freedom; the corresponding likelihood-ratio <italic>&#x03C7;</italic><sup>2</sup> statistic is reported alongside each <italic>P</italic> value. The 95% CIs for odds ratios (ORs) were obtained using profile likelihood. We used 30-node Gauss-Hermite integration over the fitted physician and case random-effect distributions to obtain population-averaged marginal probabilities, the absolute risk difference (ARD), risk ratio (RR), simulation-context number needed to treat equivalent (NNT-equivalent), and relative error-rate reduction (RER). To preserve clinical readability, the complete mathematical formulation and execution code are available in our public repository [<xref ref-type="bibr" rid="ref27">27</xref>]. RER was defined as 1 minus the ratio of error rate in the AI-assisted condition over that in the unassisted condition. Because this was a simulation-based physician performance study, the NNT-equivalent should be interpreted as a contextual measure per simulated consultation, not as an estimate of patient-level clinical benefit. Crude Top-3 diagnostic accuracy was summarized using counts and percentages. Percentile 95% CIs for the marginal probabilities, ARD, RR, and RER were obtained from 10,000 parametric bootstrap replicates that simulated new outcomes and random effects and refitted the complete model. NNT-equivalent CIs were obtained by inversion of the ARD CI and were reported only when the ARD estimate and its entire CI were positive. Interrater agreement for physician- and AI-generated diagnosis adjudication was assessed using the Fleiss &#x03BA;.</p></sec><sec id="s2-9-2"><title>Secondary Safety Outcome: Concordance With Incorrect AI Suggestions</title><p>The secondary safety analysis was restricted to risk scenarios, defined as consultations in which the AI&#x2019;s operative diagnostic suggestion was incorrect. Crude arm-specific concordance proportions within risk scenarios were reported with Wilson 95% CIs. A mixed effects logistic regression model estimated the odds of concordance with an incorrect operative AI suggestion, with interpretation differing by condition: adoption of an incorrect displayed suggestion in the AI-assisted condition and error concordance with an incorrect background suggestion in the unassisted simulation condition. Model-adjusted concordance probabilities and the adjusted absolute difference were derived using the same Gauss-Hermite integration and parametric-bootstrap procedure as for the primary outcome; concordance with correct and incorrect operative AI suggestions across all consultations was summarized descriptively by condition.</p></sec><sec id="s2-9-3"><title>Formative Workflow and Acceptability Measures</title><p>Consultation duration was summarized using means and medians with IQRs, and the relative difference in mean duration between conditions was calculated. Differences in mean consultation duration were assessed using Welch <italic>t</italic> tests, with degrees of freedom calculated using the Welch-Satterthwaite approximation. Physician-reported usefulness and satisfaction were summarized descriptively using means and SDs. To assess whether the association between study arm and diagnostic accuracy remained after accounting for consultation duration, the primary mixed effects model was refitted with log-transformed, standardized consultation duration as an additional fixed effect.</p></sec><sec id="s2-9-4"><title>Sensitivity and Exploratory Analyses</title><p>To assess robustness to the achieved physician-level allocation imbalance, we refitted the primary model in leave-one-physician-out sensitivity analyses. For each refitted model, the ARD, RR, and NNT-equivalent were recalculated. The post hoc Top-1 and Top-2 sensitivity analyses used the same mixed effects logistic regression structure as the primary analysis, with Top-1 and Top-2 correctness as binary outcomes. Model-adjusted marginal probabilities and ARDs for Top-1 and Top-2 accuracy were obtained using the same Gauss-Hermite integration and 10,000-replicate parametric-bootstrap procedure as for the primary outcome. In a separate post hoc sensitivity analysis, we refitted the primary model after excluding all consultations with quality performance issues. These issues were identified through a retrospective audit of the patient simulator transcripts. A consultation was flagged when the retrospective audit identified a simulator deviation in any assessed performance domain; all flagged consultations were excluded in this sensitivity refit.</p><p>To explore whether the intervention&#x2019;s effect was moderated by case difficulty or physician ability, we conducted 2 separate analyses. We added an interaction term between the study arm and the post hoc case difficulty or physician ability to the primary mixed effects model. Interaction tests used the continuous case-difficulty and physician-ability metrics. For presentation, consultations were grouped into quartiles of case difficulty, whereas physicians were grouped into quartiles of their unassisted Top-3 accuracy; model-adjusted probabilities and ARDs were estimated within each quartile using the corresponding interaction model, and NNT-equivalents were reported only when the ARD estimate and its entire 95% CI were positive. Because case difficulty and physician ability were derived from unassisted condition performance, these moderation analyses were considered exploratory and hypothesis-generating rather than independent subgroup validations. Recent workload was defined as the number of consultations completed by the same physician during the preceding 2 hours, and case position was defined as the consultation&#x2019;s ordinal position among the physician&#x2019;s 20 cases. Both variables were standardized before analysis. Their effects were evaluated in an exploratory mixed effects model including recent workload, case position, study arm, and the corresponding interactions with study arm, using the same physician and case random-intercept structure. The exploratory comparison with standalone AI used a mixed effects logistic regression model including evaluator type (physician vs standalone AI), study arm, and their interaction, with a case random intercept and a physician-specific random effect applied only to physician observations.</p><p>Data processing and descriptive analyses were performed in Python 3.11 using pandas 2.2.3, statsmodels 0.14.4, and SciPy 1.15.2. The GLMMs were fitted in R 4.6.1 with lme4 2.0&#x2010;6, invoked noninteractively from Python through Rscript.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Participant and Case Characteristics</title><p>The simulation was conducted between April 1, 2025, and May 20, 2025, during which 13 board-certified family and community medicine physicians from the Madrid regional health service (SERMAS) each evaluated 20 simulated patients through the web-based virtual clinic. Once all simulations were completed, the 3 independent adjudicators evaluated the correctness of the diagnoses.</p><p>Overall, 140 episodes were randomly assigned to the AI-assisted condition, and 120 episodes were randomly assigned to the unassisted simulation condition. This imbalance reflected chance variation from case-level randomization without blocking or stratification. All randomized consultations generated a submitted diagnosis; no consultation required repetition because of a reported technical failure. No episodes were lost or excluded after randomization, and all 260 randomized consultations were adjudicated and analyzed according to assigned condition (<xref ref-type="fig" rid="figure3">Figure 3</xref>). The baseline characteristics of the simulated patients are shown in <xref ref-type="table" rid="table1">Table 1</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Study flow diagram showing that all 260 randomized consultations were completed and analyzed without any loss reported or detected. AI: artificial intelligence.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e104579_fig03.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Baseline characteristics of simulated cases by study condition.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Unassisted condition (n=120), n (%)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="bottom">AI<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>-assisted condition (n=140), n (%)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Sex</td></tr><tr><td align="left" valign="top">&#x2003;Women</td><td align="left" valign="top">&#x202F;54&#x202F;(45)</td><td align="left" valign="top">&#x202F;70&#x202F;(50)</td></tr><tr><td align="left" valign="top">&#x2003;Men</td><td align="left" valign="top">&#x202F;66&#x202F;(55)</td><td align="left" valign="top">&#x202F;70&#x202F;(50)</td></tr><tr><td align="left" valign="top" colspan="3">Age group (years)</td></tr><tr><td align="left" valign="top">&#x2003;18&#x2010;30</td><td align="left" valign="top">&#x202F;28&#x202F;(23.4)</td><td align="left" valign="top">&#x202F;32&#x202F;(22.9)</td></tr><tr><td align="left" valign="top">&#x2003;31&#x2010;50 &#x202F;years</td><td align="left" valign="top">&#x202F;46&#x202F;(38.3)</td><td align="left" valign="top">&#x202F;50&#x202F;(35.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x003E;50 &#x202F;years</td><td align="left" valign="top">&#x202F;46&#x202F;(38.3)</td><td align="left" valign="top">&#x202F;58&#x202F;(41.4)</td></tr><tr><td align="left" valign="top" colspan="3">Comorbidities</td></tr><tr><td align="left" valign="top">&#x2003;With comorbidities</td><td align="left" valign="top">&#x202F;67&#x202F;(55.8)</td><td align="left" valign="top">86&#x202F;(61.4)</td></tr><tr><td align="left" valign="top">&#x2003;Without comorbidities</td><td align="left" valign="top">53 (44.2)</td><td align="left" valign="top">54 (38.6)</td></tr><tr><td align="left" valign="top" colspan="3">Diagnostic category</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gastrointestinal</td><td align="left" valign="top">&#x202F;20&#x202F;(16.7)</td><td align="left" valign="top">&#x202F;18&#x202F;(12.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Endocrinological</td><td align="left" valign="top">&#x202F;16&#x202F;(13.3)</td><td align="left" valign="top">&#x202F;28&#x202F;(20)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neurological&#x202F;and&#x202F;neuromuscular</td><td align="left" valign="top">&#x202F;15&#x202F;(12.5)</td><td align="left" valign="top">&#x202F;16&#x202F;(11.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cardiovascular&#x202F;and&#x202F;vascular</td><td align="left" valign="top">&#x202F;14&#x202F;(11.7)</td><td align="left" valign="top">&#x202F;16&#x202F;(11.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Infectious</td><td align="left" valign="top">&#x202F;13&#x202F;(10.8)</td><td align="left" valign="top">&#x202F;18&#x202F;(12.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Systemic autoimmune</td><td align="left" valign="top">&#x202F;12&#x202F;(10)</td><td align="left" valign="top">&#x202F;15&#x202F;(10.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hematologic&#x202F;and&#x202F;oncologic</td><td align="left" valign="top">&#x202F;9&#x202F;(7.5)</td><td align="left" valign="top">&#x202F;12&#x202F;(8.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Genetic/respiratory</td><td align="left" valign="top">&#x202F;7&#x202F;(5.8)</td><td align="left" valign="top">&#x202F;2&#x202F;(1.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Renal</td><td align="left" valign="top">&#x202F;6&#x202F;(5)</td><td align="left" valign="top">&#x202F;7&#x202F;(5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Musculoskeletal</td><td align="left" valign="top">&#x202F;5&#x202F;(4.2)</td><td align="left" valign="top">&#x202F;2&#x202F;(1.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Obstetric&#x202F;and&#x202F;gynecologic</td><td align="left" valign="top">&#x202F;3&#x202F;(2.5)</td><td align="left" valign="top">&#x202F;6&#x202F;(4.3)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Percentages are column percentages within each category.</p></fn><fn id="table1fn2"><p><sup>b</sup>AI: artificial intelligence.</p></fn></table-wrap-foot></table-wrap><p>Among the 13 participating physicians, 8 were women (62%) and 5 were men (38%); median age was 30.0 years (IQR 28.0&#x2010;43.0). Participants had a median of 6.0 (IQR 4.0&#x2010;19.0) years of clinical experience. The distribution of cases among physicians is given in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Distribution of cases per physician and study condition.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Physician</td><td align="left" valign="bottom">Unassisted condition (n=120), n (%)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="bottom">AI<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>-assisted condition (n=140), n (%)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">1</td><td align="left" valign="top">11 (55)</td><td align="left" valign="top">9 (45)</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">7 (35)</td><td align="left" valign="top">13 (65)</td></tr><tr><td align="left" valign="top">3</td><td align="left" valign="top">7 (35)</td><td align="left" valign="top">13 (65)</td></tr><tr><td align="left" valign="top">4</td><td align="left" valign="top">12 (60)</td><td align="left" valign="top">8 (40)</td></tr><tr><td align="left" valign="top">5</td><td align="left" valign="top">13 (65)</td><td align="left" valign="top">7 (35)</td></tr><tr><td align="left" valign="top">6</td><td align="left" valign="top">5 (25)</td><td align="left" valign="top">15 (75)</td></tr><tr><td align="left" valign="top">7</td><td align="left" valign="top">9 (45)</td><td align="left" valign="top">11 (55)</td></tr><tr><td align="left" valign="top">8</td><td align="left" valign="top">6 (30)</td><td align="left" valign="top">14 (70)</td></tr><tr><td align="left" valign="top">9</td><td align="left" valign="top">9 (45)</td><td align="left" valign="top">11 (55)</td></tr><tr><td align="left" valign="top">10</td><td align="left" valign="top">10 (50)</td><td align="left" valign="top">10 (50)</td></tr><tr><td align="left" valign="top">11</td><td align="left" valign="top">9 (45)</td><td align="left" valign="top">11 (55)</td></tr><tr><td align="left" valign="top">12</td><td align="left" valign="top">12 (60)</td><td align="left" valign="top">8 (40)</td></tr><tr><td align="left" valign="top">13</td><td align="left" valign="top">10 (50)</td><td align="left" valign="top">10 (50)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Each physician completed 20 consultations. Percentages are row percentages within physician.</p></fn><fn id="table2fn2"><p><sup>b</sup>AI: artificial intelligence.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Outcomes</title><sec id="s3-2-1"><title>Primary Outcome: Top-3 Diagnostic Accuracy</title><p>Diagnoses provided by physicians and those generated by the AI assistance were reviewed and evaluated by 3 independent adjudicators. Interrater reliability among the 3 adjudicators, measured using the Fleiss &#x03BA;, was &#x03BA;=0.902 (95% CI 0.864 to 0.937) for diagnoses provided by the physicians and &#x03BA;=0.896 (95% CI 0.858 to 0.934) for AI-generated diagnoses. The majority rule yielded a final classification for every adjudicated assessment; therefore, no assessment was discarded.</p><p>Physicians using AI assistance achieved higher diagnostic accuracy for the primary outcome. A correct diagnosis was provided in 78 of 120 consultations (65%) in the unassisted condition and 105 of 140 consultations (75%) in the AI-assisted condition. In the mixed effects model, adjusted diagnostic accuracy was 62.3% (95% CI 49.1% to 75.7%) in the unassisted condition and 74.6% (95% CI 63.7% to 85.4%) in the AI-assisted condition. Assignment to the AI-assisted condition was associated with higher Top-3 diagnostic accuracy (AOR 2.68, 95% CI 1.24 to 6.12; &#x03C7;<sup>2</sup><sub>1</sub>=6.4, <italic>P</italic>=.01), corresponding to an adjusted absolute difference of +12.3 (95% CI +2.7 to +22.6) percentage points, a simulation-context NNT-equivalent of 8.1 (95% CI 4.4 to 37.2), and RER of 32.6% (95% CI 8.9% to 54.4%).</p></sec><sec id="s3-2-2"><title>Secondary Safety Outcome: Concordance With Incorrect AI Suggestions</title><p>For the secondary safety analysis, risk scenarios were defined as consultations in which the AI&#x2019;s operative diagnostic suggestion was incorrect. In the AI-assisted condition, physicians saw the AI suggestions during the consultation; therefore, a physician diagnostic error concordant with an incorrect displayed suggestion was interpreted as adoption of that suggestion and as an overreliance signal. In the unassisted simulation condition, the AI suggestion was generated only in the background and was not displayed to physicians; therefore, physician error concordant with an incorrect background suggestion was interpreted as error concordance rather than true adoption. In the AI-assisted condition, 51 of 140 consultations were risk scenarios. Physicians submitted a first diagnosis concordant with the incorrect displayed AI suggestion in 29 of these 51 consultations (56.9%, 95% CI 43.3% to 69.5%). In the unassisted simulation condition, 47 of 120 consultations were risk scenarios based on the background AI suggestion, and physicians submitted a first diagnosis concordant with an erroneous AI suggestion in 19 of these 47 consultations (40.4%, 95% CI 27.6% to 54.7%). In the mixed effects model restricted to risk scenarios, the point estimate for concordance was higher in the AI-assisted condition, but the comparison was not statistically significant (AOR 2.08, 95% CI 0.89 to 5.19; &#x03C7;<sup>2</sup><sub>1</sub>=2.9, <italic>P</italic>=.09). Adjusted concordance probabilities were 55.8% in the AI-assisted condition and 38.8% in the unassisted condition, corresponding to an adjusted absolute difference of +17.0 percentage points (95% CI &#x2013;2.7 to +36.8).</p><p>When viewed across all consultations, concordance with an incorrect operative AI suggestion occurred in 29 of 140 AI-assisted consultations (20.7%) and in 19 of 120 unassisted consultations (15.8%). Conversely, concordance with a correct operative AI suggestion occurred in 52.1% (73/140) of AI-assisted consultations and 39.2% (47/120) of unassisted consultations. These cohort-wide rates should be interpreted as descriptive measures of concordance with AI outputs; only in the AI-assisted condition can concordance with a displayed incorrect suggestion be interpreted as possible adoption.</p></sec><sec id="s3-2-3"><title>Formative Workflow and Acceptability Measures</title><p>As part of the formative evaluation, we also assessed workflow burden and physician-reported acceptability. Mean consultation time was 328.64 seconds in the unassisted condition and 363.72 seconds in the AI-assisted condition (<italic>t</italic><sub>257.6</sub>=2.11, <italic>P</italic>=.04), corresponding to a 10.7% increase. Median consultation time was 308.86 (IQR 243.39&#x2010;383.17) seconds in the unassisted condition and 341.67 (IQR 278.50&#x2010;428.99) seconds in the AI-assisted condition. After adding log-transformed, standardized consultation duration to the primary mixed effects model, the association with study arm remained statistically significant (&#x03C7;<sup>2</sup><sub>1</sub>=6.0, <italic>P</italic>=.01), whereas consultation duration was not independently associated with diagnostic accuracy (&#x03C7;<sup>2</sup><sub>1</sub>=1.7, <italic>P</italic>=.19).</p><p>Physicians rated the evaluated system highly, with a mean score of 4.38 (SD 0.47) for usefulness and 4.46 (SD 0.41) for overall satisfaction on a 5-point scale.</p></sec><sec id="s3-2-4"><title>Sensitivity and Exploratory Analyses</title><p>In the post hoc Top-1 sensitivity analysis, correct Top-1 diagnoses occurred in 61 of 120 unassisted consultations (50.8%) and 79 of 140 AI-assisted consultations (56.4%). Top-1 point estimates favored the AI-assisted condition, but the comparison was not statistically significant (AOR 1.83, 95% CI 0.93 to 3.80; &#x03C7;<sup>2</sup><sub>1</sub>=3.1, <italic>P</italic>=.08). Adjusted probabilities were 47.8% in the unassisted condition and 57% in the AI-assisted condition, corresponding to an adjusted absolute difference of +9.2 percentage points (95% CI &#x2013;1.0 to +20.1). In the corresponding post hoc Top-2 sensitivity analysis, correct Top-2 diagnoses occurred in 74 of 120 unassisted consultations (61.7%) and 100 of 140 AI-assisted consultations (71.4%). The AI-assisted condition was associated with higher Top-2 diagnostic accuracy (AOR 2.59, 95% CI 1.23 to 5.77; &#x03C7;<sup>2</sup><sub>1</sub>=6.3, <italic>P</italic>=.01), with adjusted probabilities of 57.9% and 70.7%, respectively, corresponding to an adjusted absolute difference of +12.8 percentage points (95% CI +3.1 to +23.2). Because these analyses were post hoc, Top-2 was interpreted as supportive sensitivity evidence, whereas Top-1 was directionally favorable but inconclusive.</p><p>In leave-one-physician-out sensitivity analyses, the direction and magnitude of the primary effect remained stable. Across the 13 refitted models, the adjusted absolute difference ranged from +9.2 to +14.3 percentage points, the adjusted risk ratio ranged from 1.14 to 1.24, and the simulation-context NNT-equivalent ranged from 7.0 to 10.9.</p><p>As a result of the retrospective virtual patient performance audit, 16 of 260 consultations were flagged for virtual patient quality deviations related to adherence to role-play instructions. These consultations were excluded in a sensitivity analysis that was done over the remaining 244 consultations. The primary diagnostic effect remained consistent: The AI-assisted condition was associated with higher diagnostic accuracy (AOR 2.78, 95% CI 1.28 to 6.41; &#x03C7;<sup>2</sup><sub>1</sub>=6.8, <italic>P</italic>=.009), with adjusted accuracies of 61.4% in the unassisted condition and 74.6% in the AI-assisted condition, corresponding to an adjusted absolute difference of +13.2 percentage points (95% CI +3.3 to +23.6) and a simulation-context NNT-equivalent of 7.6 (95% CI 4.2 to 30.2).</p><p>We computed case difficulty for every case. The weighted mean case difficulty score was similar in the unassisted condition (0.3500, SD 0.3605) and AI-assisted condition (0.3575, SD 0.3716) groups. Two consultations from one case without any unassisted observations were excluded because case difficulty could not be derived for that case. In an exploratory interaction analysis, the association between AI assistance and diagnostic accuracy varied significantly with case difficulty (interaction &#x03C7;<sup>2</sup><sub>1</sub>=18.4, <italic>P</italic>&#x003C;.001). Quartile-specific adjusted accuracies and adjusted absolute differences are shown in <xref ref-type="fig" rid="figure4">Figure 4</xref>. The absolute difference was +43.9 percentage points (95% CI +22.8 to +64.3) in the highest-difficulty quartile and +31.1 points (95% CI +15.5 to +48.9) in the quartile 2. No clear difference was observed in quartile 3 (+0.6 points, 95% CI &#x2013;11.2 to +12.2), whereas the least difficult quartile favored the unassisted condition (&#x2013;7.7 points, 95% CI &#x2013;16.3 to &#x2013;0.6). The simulation-context NNT-equivalent was 2.3 (95% CI 1.6 to 4.4) in the highest-difficulty quartile and 3.2 (95% CI 2.0 to 6.5) in quartile 2; it was not reported for quartiles 3 or 4 because the ARD CI crossed zero or the ARD point estimate was negative.</p><p>In a second exploratory analysis, we examined the interaction between study arm and physician ability. The interaction was statistically significant (&#x03C7;<sup>2</sup><sub>1</sub>=11.4, <italic>P</italic>&#x003C;.001), with larger adjusted differences in the lower unassisted-ability strata. The adjusted absolute difference was +34.8 percentage points (95% CI +19.5 to +51.6) in the lowest unassisted accuracy quartile and +10.7 points (95% CI +1.7 to +20.3) in quartile 2; the CIs included no difference in quartile 3 (+2.4 points, 95% CI &#x2013;7.8 to +12.7) and quartile 4 (&#x2013;3.0 points, 95% CI &#x2013;14.6 to +8.4). Because physician ability was derived from unassisted performance and the physician-level variance was estimated at the boundary in this model, these findings were treated as hypothesis-generating.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Adjusted diagnostic accuracy and absolute risk difference (ARD) by case difficulty quartile (Q) and overall. Bars show model-adjusted diagnostic accuracy (%) in the unassisted and artificial intelligence (AI)-assisted conditions. Triangles show adjusted absolute risk differences in percentage points, and circles with dashed lines show the corresponding 95% CI bounds.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e104579_fig04.png"/></fig><p>To assess the influence of workload (the number of cases solved in the preceding 2 h) and consultation order (position among the 20 cases for each physician), we fitted an additional exploratory mixed effects model including standardized recent workload, standardized case position, and their interactions with study arm (<xref ref-type="table" rid="table3">Table 3</xref>). Point estimates were directionally consistent with a smaller AI effect at higher recent workload and a larger effect later in the sequence, but neither interaction reached statistical significance (<xref ref-type="table" rid="table3">Table 3</xref>).</p><p>To explore how physician performance compared with standalone AI performance, we used a mixed effects logistic regression model including evaluator type (physician vs standalone AI), study arm, and their interaction. In the unassisted condition, the physician versus standalone-AI contrast was not statistically significant (OR 0.56, 95% CI 0.20 to 1.59; &#x03C7;<sup>2</sup><sub>1</sub>=1.2, <italic>P</italic>=.27). The human-by-arm interaction was also not statistically significant (OR 2.21, 95% CI 0.68 to 7.35; &#x03C7;<sup>2</sup><sub>1</sub>=1.7, <italic>P</italic>=.19), providing no evidence that the physician versus AI contrast differed between conditions. In the AI-assisted condition, physician performance did not differ significantly from standalone AI performance (OR 1.24, 95% CI 0.45 to 3.44; &#x03C7;<sup>2</sup><sub>1</sub>=0.2, <italic>P</italic>=.67). These exploratory findings do not establish additive human-AI synergy.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Exploratory arm-by-workload and arm-by-case-position interaction estimates, with workload and case position standardized prior to analysis.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Interaction</td><td align="left" valign="bottom">Coefficient</td><td align="left" valign="bottom">OR (profile-likelihood 95% CI)</td><td align="left" valign="bottom">Likelihood-ratio &#x03C7;<sup>2</sup> (degrees of freedom)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom"><italic>P</italic> value<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Study arm &#x00D7; workload</td><td align="left" valign="top">&#x2212;0.787</td><td align="left" valign="top">0.46 (0.18&#x2010;1.10)</td><td align="left" valign="top">3.0 (1)</td><td align="left" valign="top">.08</td></tr><tr><td align="left" valign="top">Study arm &#x00D7; case position</td><td align="left" valign="top">0.720</td><td align="left" valign="top">2.05 (0.82&#x2010;5.40)</td><td align="left" valign="top">2.4 (1)</td><td align="left" valign="top">.12</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup><italic>&#x03C7;</italic><sup>2</sup> statistics and <italic>P</italic> values were obtained using likelihood-ratio tests against the corresponding nested models.</p></fn></table-wrap-foot></table-wrap></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this high-difficulty, simulated primary care setting, real-time access to the AI diagnostic assistant was associated with higher Top-3 diagnostic accuracy than an unassisted, resource-restricted simulation condition. This finding was stable in leave-one-physician-out and virtual patient simulator exclusion analyses; the post hoc Top-2 analysis was supportive, whereas Top-1 was directionally favorable but statistically inconclusive. The study was designed as a diagnostic stress test rather than as an estimate of routine clinical effectiveness; therefore, absolute effect measures, including the simulation-context NNT-equivalent, should be interpreted within the curated challenging case mix.</p><p>The study also showed a possible safety-relevant overreliance signal when incorrect AI suggestions were visible, although the adjusted between condition comparison was imprecise and did not reach statistical significance. In the AI-assisted condition, concordance with an incorrect displayed suggestion can reasonably be interpreted as possible adoption of that suggestion. In the unassisted condition, however, concordance with a background AI suggestion reflects error concordance rather than true adoption, because physicians did not see the AI output.</p><p>Finally, exploratory analyses suggested marked heterogeneity, with benefit concentrated in more difficult cases and lower unassisted accuracy physician strata, while the easiest case quartile favored the unassisted condition. However, these subgroup findings should be considered hypothesis-generating because they were exploratory and partly derived from unassisted-condition performance.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>These findings extend prior studies of LLM-based diagnostic support in 2 ways. First, most randomized evidence has evaluated static text vignettes or structured patient care tasks rather than real-time, voice-based support during an unfolding simulated consultation. In a vignette-based, randomized study, access to GPT-4 did not significantly improve physicians&#x2019; diagnostic reasoning compared with conventional resources, even though the standalone model performed well [<xref ref-type="bibr" rid="ref9">9</xref>]. A later randomized trial of GPT-4 assistance in patient care tasks found a more modest physician-performance improvement in a sequential information environment [<xref ref-type="bibr" rid="ref10">10</xref>]. Other recent work suggests that stronger benefits may occur when LLM support is paired with AI-literacy training or embedded in collaborative diagnostic workflows [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>Second, this study evaluated an interactive voice-based workflow in which AI suggestions were generated during the consultation while physicians were still gathering information and forming hypotheses. This differs from benchmark studies showing strong standalone LLM performance on curated cases [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>] and from evaluations focused mainly on passive documentation or operational efficiency. These results therefore support the view that the clinical value of diagnostic AI may depend not only on model capability but also on how model outputs are integrated into physicians&#x2019; real-time reasoning workflow.</p><p>The overreliance signal is also consistent with prior human-AI interaction research. Studies in digital pathology, radiology, and LLM-assisted diagnostic reasoning have shown that correct AI outputs can improve clinician performance, whereas incorrect outputs can degrade it [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. More broadly, automation bias research has long shown that users may overweight automated recommendations in high-stakes environments, especially when systems appear authoritative or are presented as decision-support tools [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. The higher crude concordance with incorrect suggestions in the AI-assisted condition was consistent with a possible overreliance concern in this real-time simulated primary care context, although the adjusted between-condition comparison was not statistically significant.</p></sec><sec id="s4-3"><title>Implications for Human-AI Diagnostic Support</title><p>The main implication is not that real-time diagnostic AI assistance is ready for routine clinical deployment but that this type of system merits further formative refinement and prospective evaluation. In this controlled simulation, AI assistance appeared to support diagnostic performance. However, point estimates suggested a possible overreliance risk when the displayed suggestion was wrong, although the between-condition comparison was inconclusive. Future systems should therefore be evaluated not only for average diagnostic accuracy but also for how they behave under incorrect-suggestion scenarios and how physicians respond to those errors.</p><p>The results also suggest several design priorities. Interfaces should preserve physician verification rather than simply making AI suggestions more salient. Potential mitigation strategies include clearer uncertainty displays, prompts that ask clinicians to consider alternatives before accepting a suggestion, explicit flags when evidence is incomplete, and interface designs that separate hypothesis generation from recommendation endorsement. Training may also be important, because recent randomized evidence suggests that physicians may benefit more from LLM support when they receive AI-literacy training or when collaboration is structured through dedicated workflows [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>Finally, the exploratory comparison with standalone AI should be interpreted cautiously. Neither physician-versus-AI contrast nor the human-by-arm interaction reached statistical significance, so these data do not establish additive human-AI synergy or a difference in relative performance between conditions. Whether specific workflows can produce reliable additive benefit beyond standalone model performance remains an empirical question for future studies.</p></sec><sec id="s4-4"><title>Strengths and Limitations</title><p>The strengths of this study include case-level randomization, complete inclusion of all randomized consultations, blinded adjudication by 3 independent physicians, excellent interrater reliability, and mixed effects modeling with random intercepts for physician and case. The architecture also separated the patient simulator from the AI assistant, preventing the evaluated assistant from accessing the underlying vignette text or simulator prompts. The use of voice-based virtual consultations represents a more interactive evaluation setting than static text vignettes while retaining the safety and reproducibility advantages of simulation.</p><p>The study also has important limitations. First, this was a simulation-based physician performance study, not a clinical trial involving real patients or patient outcomes. Virtual patients cannot fully reproduce pain, anxiety, memory limitations, nonverbal cues, physical examination ambiguity, workflow interruptions, or the consequences of live clinical decisions. The retrospective assessment of the simulated patient performance showed imperfect behavior, which does not invalidate the findings, as demonstrated by the sensitivity analysis, but does underscore the inherently artificial nature of the environment.</p><p>Second, the comparator was an unassisted, resource-restricted simulation condition rather than resource-enabled usual care. Although this design standardized the experimental contrast, physicians in routine primary care may consult point-of-care references, clinical guidelines, internet searches, colleagues, or follow-up information. Access to such resources could improve unassisted performance and attenuate the observed relative and absolute contrast. However, the proportion of the observed effect attributable to the resource restriction cannot be estimated from this 2-condition design, because conventional resources could affect both unassisted and AI-assisted performance and may interact with AI use. Accordingly, the reported effect estimates apply specifically to AI assistance versus the resource-restricted comparator and should not be interpreted as the incremental effect of AI over resource-enabled usual care. Future studies should include a conventional resource comparator under the same time constraints and make the same resources available alongside AI assistance. This limitation is especially relevant because the case mix was intentionally difficult. As a result, the observed absolute accuracy gain and simulation-context NNT-equivalent may overestimate the absolute benefit that would be observed in routine care with easier cases and broader diagnostic resources.</p><p>Third, the physician sample was modest and recruited from a single Spanish regional health service. Although leave-one-physician-out analyses suggested that the primary result was not driven by a single physician, they address robustness to individual participants rather than external validity. The localized sample limits generalizability to physicians with different training backgrounds, experience, digital familiarity, health care systems, and practice environments. Because participation required use of a web application and voice interface, recruitment may also have favored physicians with greater baseline digital comfort. Future studies should recruit larger physician samples across multiple regions and health care systems and prospectively characterize participants&#x2019; digital literacy. The achieved allocation was also imbalanced because randomization was not blocked or stratified by physician or case, although the primary model included physician and case random intercepts and the leave-one-physician-out analyses supported robustness.</p><p>Fourth, several analyses were exploratory. Case difficulty and physician ability were derived from unassisted-condition performance, which introduces potential mathematical coupling when those metrics are reintroduced into the models. The workload, case position, physician ability, case difficulty, and standalone AI comparisons should therefore be treated as hypothesis-generating rather than confirmatory. The physician ability and safety models also yielded boundary estimates for the physician-level random effect variance, further supporting cautious interpretation.</p><p>Finally, the safety analysis used concordance with incorrect AI suggestions as an operational marker. In the AI-assisted condition, this is a plausible measure of possible overreliance because suggestions were visible. In the unassisted condition, however, the same measure represents background error concordance rather than true adoption. The adjusted 95% CI included no between-condition difference. Accordingly, the between-condition difference should be interpreted as a safety-relevant concordance signal rather than as a clinical harm estimate.</p></sec><sec id="s4-5"><title>Conclusions</title><p>In this randomized simulation study, real-time, voice-based AI assistance was associated with higher Top-3 diagnostic accuracy than an unassisted, resource-restricted simulation condition. The study also identified a possible overreliance signal when incorrect AI suggestions were visible, although the between-condition safety estimate was imprecise and did not reach statistical significance. These findings support further refinement of diagnostic AI interfaces and prospective evaluation in real clinical workflows before any conclusions are drawn about routine clinical effectiveness or implementation.</p></sec></sec></body><back><ack><p>We would like to express our gratitude to the 13 primary care physicians who participated in the study and to the 3 independent adjudicators for their meticulous work in evaluating the diagnoses. We thank Akira Vleming for contributing to the grammar review and preparation of figures. We also thank Jos&#x00E9; Ignacio Klett Mingo and Julio Cambronero Plaza of Synthetrial and Laura Mu&#x00F1;oz Ortiz of Datexbio for their valuable independent contributions to the review and refinement of the study&#x2019;s statistical methods and analyses. During the preparation of this manuscript, the authors utilized the generative artificial intelligence (AI) language models Gemini from Google and ChatGPT from OpenAI and AI-based writing support tool Grammarly. Their use was limited to assisting with brainstorming ideas, refining sentence structure for clarity, and identifying appropriate terminology. All AI-generated suggestions were critically reviewed, edited, and validated by the authors, who retain full responsibility for the accuracy, integrity, and final content of this manuscript.</p></ack><notes><sec><title>Funding</title><p>This study was funded by Medsys AI, SL, which also provided access to the Medsys AI system evaluated in the study. The funder had no role in physician recruitment, data collection, blinded adjudication of diagnostic correctness, or the final decision to submit the manuscript for publication. MP&#x2019;s role as a shareholder of Medsys AI, SL, and his contributions to the study are disclosed in the Conflicts of Interest and Authors&#x2019; Contributions sections.</p></sec><sec><title>Data Availability</title><p>The full dataset and all materials supporting the findings of this study are available in a public repository [<xref ref-type="bibr" rid="ref27">27</xref>]. The repository provides unrestricted access to the complete study protocol, the 40 clinical vignettes used, the dataset derived from all 260 virtual consultations, the evaluation by the adjudicators, the statistical analysis code, complete outputs from the full and simulator-exclusion analyses, the simulator deviation analysis, the analytical transparency document, and a data dictionary. To protect participant privacy, physician names and directly identifying information are not included in the public repository. Access to any additional nonpublic participant identifying information, if justified, may be requested by qualified researchers by contacting the corresponding author, subject to a data access agreement.</p></sec></notes><fn-group><fn fn-type="con"><p>IC and MP contributed to the conceptualization and design of the study. IC and RGC were responsible for data collection and curation. All three authors (IC, MP, and RGC) participated in the statistical analysis and in writing the original draft and subsequent revisions of the report. All authors confirm they had full access to all the data in the study and accept final responsibility for the decision to submit for publication. IC and RGC have accessed and verified the underlying data reported in the manuscript.</p></fn><fn fn-type="conflict"><p>MP is a shareholder of Medsys AI, SL, the company that developed the Medsys AI system evaluated in this study and funded the study. MP contributed to the conceptualization and design of the study, statistical analysis, and manuscript preparation, as described in the Author Contributions section. To mitigate potential bias, diagnostic correctness was assessed by 3 independent blinded adjudicators, analyses were conducted on a locked dataset derived from their adjudications, and the full dataset and statistical analysis code were made publicly available to support transparency and third-party verification. IC and RGC declare no competing interests.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AI</term><def><p>artificial intelligence</p></def></def-item><def-item><term id="abb2">AOR</term><def><p>adjusted odds ratio</p></def></def-item><def-item><term id="abb3">ARD</term><def><p>absolute risk difference</p></def></def-item><def-item><term id="abb4">GLMM</term><def><p>generalized linear mixed model</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">NNT</term><def><p>number needed to treat</p></def></def-item><def-item><term id="abb7">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb8">RER</term><def><p>relative error-rate reduction</p></def></def-item><def-item><term id="abb9">RR</term><def><p>risk ratio</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Meyer</surname><given-names>AND</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>EJ</given-names> </name></person-group><article-title>The frequency of diagnostic errors in outpatient care: estimations from three large observational studies involving US adult populations</article-title><source>BMJ Qual Saf</source><year>2014</year><month>09</month><volume>23</volume><issue>9</issue><fpage>727</fpage><lpage>731</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2013-002627</pub-id><pub-id pub-id-type="medline">24742777</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Schiff</surname><given-names>GD</given-names> </name><name name-style="western"><surname>Graber</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Onakpoya</surname><given-names>I</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>MJ</given-names> </name></person-group><article-title>The global burden of diagnostic errors in primary care</article-title><source>BMJ Qual Saf</source><year>2017</year><month>06</month><volume>26</volume><issue>6</issue><fpage>484</fpage><lpage>494</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2016-005401</pub-id><pub-id pub-id-type="medline">27530239</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kostopoulou</surname><given-names>O</given-names> </name><name name-style="western"><surname>Delaney</surname><given-names>BC</given-names> </name><name name-style="western"><surname>Munro</surname><given-names>CW</given-names> </name></person-group><article-title>Diagnostic difficulty and error in primary care--a systematic review</article-title><source>Fam Pract</source><year>2008</year><month>12</month><volume>25</volume><issue>6</issue><fpage>400</fpage><lpage>413</lpage><pub-id pub-id-type="doi">10.1093/fampra/cmn071</pub-id><pub-id pub-id-type="medline">18842618</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hendriksen</surname><given-names>JMT</given-names> </name><name name-style="western"><surname>Koster-van Ree</surname><given-names>M</given-names> </name><name name-style="western"><surname>Morgenstern</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>Clinical characteristics associated with diagnostic delay of pulmonary embolism in primary care: a retrospective observational study</article-title><source>BMJ Open</source><year>2017</year><month>03</month><day>9</day><volume>7</volume><issue>3</issue><fpage>e012789</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2016-012789</pub-id><pub-id pub-id-type="medline">28279993</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mitchell</surname><given-names>JL</given-names> </name></person-group><article-title>Understanding the impact of delayed diagnosis and misdiagnosis of systemic lupus erythematosus (SLE)</article-title><source>J Family Med Prim Care</source><year>2024</year><month>11</month><volume>13</volume><issue>11</issue><fpage>4819</fpage><lpage>4823</lpage><pub-id pub-id-type="doi">10.4103/jfmpc.jfmpc_1177_24</pub-id><pub-id pub-id-type="medline">39722963</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Large language models for disease diagnosis: a scoping review</article-title><source>NPJ Artif Intell</source><year>2025</year><volume>1</volume><issue>1</issue><fpage>9</fpage><pub-id pub-id-type="doi">10.1038/s44387-025-00011-z</pub-id><pub-id pub-id-type="medline">40607112</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Schaekermann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Palepu</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Towards conversational diagnostic artificial intelligence</article-title><source>Nature</source><year>2025</year><month>06</month><volume>642</volume><issue>8067</issue><fpage>442</fpage><lpage>450</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08866-7</pub-id><pub-id pub-id-type="medline">40205050</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McDuff</surname><given-names>D</given-names> </name><name name-style="western"><surname>Schaekermann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Towards accurate differential diagnosis with large language models</article-title><source>Nature</source><year>2025</year><month>06</month><volume>642</volume><issue>8067</issue><fpage>451</fpage><lpage>457</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08869-4</pub-id><pub-id pub-id-type="medline">40205049</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goh</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hom</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Large language model influence on diagnostic reasoning</article-title><source>JAMA Netw Open</source><year>2024</year><month>10</month><day>1</day><volume>7</volume><issue>10</issue><fpage>e2440969</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.40969</pub-id><pub-id pub-id-type="medline">39466245</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goh</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Strong</surname><given-names>E</given-names> </name><etal/></person-group><article-title>GPT-4 assistance for improvement of physician performance on patient care tasks: a randomized controlled trial</article-title><source>Nat Med</source><year>2025</year><month>04</month><volume>31</volume><issue>4</issue><fpage>1233</fpage><lpage>1238</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03456-y</pub-id><pub-id pub-id-type="medline">39910272</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qazi</surname><given-names>IA</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>A</given-names> </name><name name-style="western"><surname>Khawaja</surname><given-names>AU</given-names> </name><name name-style="western"><surname>Akhtar</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Sheikh</surname><given-names>AZ</given-names> </name><name name-style="western"><surname>Alizai</surname><given-names>MH</given-names> </name></person-group><article-title>Large language model diagnostic assistance for physicians in a lower-middle-income country: a randomized controlled trial</article-title><source>Nat Health</source><year>2026</year><volume>1</volume><issue>2</issue><fpage>198</fpage><lpage>205</lpage><pub-id pub-id-type="doi">10.1038/s44360-025-00007-8</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Everett</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Bunning</surname><given-names>BJ</given-names> </name><name name-style="western"><surname>Jain</surname><given-names>P</given-names> </name><etal/></person-group><article-title>From tool to teammate in a randomized controlled trial of clinician-AI collaborative workflows for diagnosis</article-title><source>NPJ Digit Med</source><year>2026</year><month>03</month><day>18</day><volume>9</volume><issue>1</issue><fpage>409</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02545-1</pub-id><pub-id pub-id-type="medline">41851268</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bubeck</surname><given-names>S</given-names> </name><name name-style="western"><surname>Petro</surname><given-names>J</given-names> </name></person-group><article-title>Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine</article-title><source>N Engl J Med</source><year>2023</year><month>03</month><day>30</day><volume>388</volume><issue>13</issue><fpage>1233</fpage><lpage>1239</lpage><pub-id pub-id-type="doi">10.1056/NEJMsr2214184</pub-id><pub-id pub-id-type="medline">36988602</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zack</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lehman</surname><given-names>E</given-names> </name><name name-style="western"><surname>Suzgun</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Assessing the potential of GPT-4 to perpetuate racial and gender biases in health care: a model evaluation study</article-title><source>Lancet Digit Health</source><year>2024</year><month>01</month><volume>6</volume><issue>1</issue><fpage>e12</fpage><lpage>e22</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00225-X</pub-id><pub-id pub-id-type="medline">38123252</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidt</surname><given-names>HG</given-names> </name><name name-style="western"><surname>Rotgans</surname><given-names>JI</given-names> </name><name name-style="western"><surname>Mamede</surname><given-names>S</given-names> </name></person-group><article-title>Bias sensitivity in diagnostic decision-making: comparing ChatGPT with residents</article-title><source>J GEN INTERN MED</source><year>2025</year><month>03</month><volume>40</volume><issue>4</issue><fpage>790</fpage><lpage>795</lpage><pub-id pub-id-type="doi">10.1007/s11606-024-09177-9</pub-id><pub-id pub-id-type="medline">39511117</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kiani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Uyumazturk</surname><given-names>B</given-names> </name><name name-style="western"><surname>Rajpurkar</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Impact of a deep learning assistant on the histopathologic classification of liver cancer</article-title><source>NPJ Digit Med</source><year>2020</year><volume>3</volume><issue>1</issue><fpage>23</fpage><pub-id pub-id-type="doi">10.1038/s41746-020-0232-8</pub-id><pub-id pub-id-type="medline">32140566</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>DY</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>AL</given-names> </name><etal/></person-group><article-title>Artificial intelligence suppression as a strategy to mitigate artificial intelligence automation bias</article-title><source>J Am Med Inform Assoc</source><year>2023</year><month>09</month><day>25</day><volume>30</volume><issue>10</issue><fpage>1684</fpage><lpage>1692</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad118</pub-id><pub-id pub-id-type="medline">37561535</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jabbour</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fouhey</surname><given-names>D</given-names> </name><name name-style="western"><surname>Shepard</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Measuring the impact of AI in the diagnosis of hospitalized patients: a randomized clinical vignette survey study</article-title><source>JAMA</source><year>2023</year><month>12</month><day>19</day><volume>330</volume><issue>23</issue><fpage>2275</fpage><lpage>2284</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.22295</pub-id><pub-id pub-id-type="medline">38112814</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gombolay</surname><given-names>GY</given-names> </name><name name-style="western"><surname>Silva</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schrum</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Effects of explainable artificial intelligence in neurology decision support</article-title><source>Ann Clin Transl Neurol</source><year>2024</year><month>05</month><volume>11</volume><issue>5</issue><fpage>1224</fpage><lpage>1235</lpage><pub-id pub-id-type="doi">10.1002/acn3.52036</pub-id><pub-id pub-id-type="medline">38581138</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abbas</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Jeong</surname><given-names>W</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SW</given-names> </name></person-group><article-title>Explainable AI in clinical decision support systems: a meta-analysis of methods, applications, and usability challenges</article-title><source>Healthcare</source><year>2025</year><volume>13</volume><issue>17</issue><fpage>2154</fpage><pub-id pub-id-type="doi">10.3390/healthcare13172154</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tierney</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Gayre</surname><given-names>G</given-names> </name><name name-style="western"><surname>Hoberman</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Ambient artificial intelligence scribes: learnings after 1 year and over 2.5 million uses</article-title><source>NEJM Catalyst</source><year>2025</year><month>04</month><day>16</day><volume>6</volume><issue>5</issue><pub-id pub-id-type="doi">10.1056/CAT.25.0040</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>SJ</given-names> </name><etal/></person-group><article-title>Ambient artificial intelligence scribes: utilization and impact on documentation time</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>02</month><day>1</day><volume>32</volume><issue>2</issue><fpage>381</fpage><lpage>385</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae304</pub-id><pub-id pub-id-type="medline">39688515</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peine</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gronholz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Seidl-Rathkopf</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Standardized comparison of voice-based information and documentation systems to established systems in intensive care: crossover study</article-title><source>JMIR Med Inform</source><year>2023</year><month>11</month><day>28</day><volume>11</volume><fpage>e44773</fpage><pub-id pub-id-type="doi">10.2196/44773</pub-id><pub-id pub-id-type="medline">38015593</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>King</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Angus</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Cooper</surname><given-names>GF</given-names> </name><etal/></person-group><article-title>A voice-based digital assistant for intelligent prompting of evidence-based practices during ICU rounds</article-title><source>J Biomed Inform</source><year>2023</year><month>10</month><volume>146</volume><fpage>104483</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2023.104483</pub-id><pub-id pub-id-type="medline">37657712</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Holderried</surname><given-names>F</given-names> </name><name name-style="western"><surname>Stegemann-Philipps</surname><given-names>C</given-names> </name><name name-style="western"><surname>Herschbach</surname><given-names>L</given-names> </name><etal/></person-group><article-title>A generative pretrained transformer (GPT)-powered chatbot as a simulated patient to practice history taking: prospective, mixed methods study</article-title><source>JMIR Med Educ</source><year>2024</year><month>01</month><day>16</day><volume>10</volume><issue>1</issue><fpage>e53961</fpage><pub-id pub-id-type="doi">10.2196/53961</pub-id><pub-id pub-id-type="medline">38227363</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Simulated patient systems powered by large language model-based AI agents offer potential for transforming medical education</article-title><source>Commun Med (Lond)</source><year>2025</year><month>12</month><day>19</day><volume>6</volume><issue>1</issue><fpage>27</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-01283-x</pub-id><pub-id pub-id-type="medline">41420084</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Cusacovich</surname><given-names>I</given-names> </name><name name-style="western"><surname>Pinilla</surname><given-names>M</given-names> </name><name name-style="western"><surname>Garcia Castro</surname><given-names>R</given-names> </name></person-group><article-title>A real-time artificial intelligence diagnostic copilot in simulated primary care consultations: randomized simulation study</article-title><source>Zenodo</source><year>2026</year><access-date>2026-09-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/22031287">https://zenodo.org/records/22031287</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="report"><person-group person-group-type="author"><collab>European Union</collab></person-group><article-title>Reglamento (UE) 2016/679 del parlamento europeo y del consejo de 27 de abril de 2016 relativo a la protecci&#x00F3;n de las personas f&#x00ED;sicas en lo que respecta al tratamiento de datos personales</article-title><year>2016</year><month>04</month><day>27</day><access-date>2026-09-10</access-date><publisher-name>Agencia Estatal Bolet&#x00ED;n Oficial del Estado</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.boe.es/doue/2016/119/L00001-00088.pdf">https://www.boe.es/doue/2016/119/L00001-00088.pdf</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="report"><person-group person-group-type="author"><collab>Kingdom of Spain</collab></person-group><article-title>Ley Org&#x00E1;nica 3/2018, de 5 de diciembre, de Protecci&#x00F3;n de Datos Personales y garant&#x00ED;a de los derechos digitales</article-title><year>2018</year><month>12</month><day>6</day><access-date>2026-09-10</access-date><publisher-name>Agencia Estatal Bolet&#x00ED;n Oficial del Estado</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.boe.es/eli/es/lo/2018/12/05/3/con">https://www.boe.es/eli/es/lo/2018/12/05/3/con</ext-link></comment></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hammoud</surname><given-names>M</given-names> </name><name name-style="western"><surname>Douglas</surname><given-names>S</given-names> </name><name name-style="western"><surname>Darmach</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alawneh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sanyal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kanbour</surname><given-names>Y</given-names> </name></person-group><article-title>Evaluating the diagnostic performance of symptom checkers: clinical vignette study</article-title><source>JMIR AI</source><year>2024</year><month>04</month><day>29</day><volume>3</volume><fpage>e46875</fpage><pub-id pub-id-type="doi">10.2196/46875</pub-id><pub-id pub-id-type="medline">38875676</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qazi</surname><given-names>IA</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>A</given-names> </name><name name-style="western"><surname>Khawaja</surname><given-names>AU</given-names> </name><name name-style="western"><surname>Akhtar</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Sheikh</surname><given-names>AZ</given-names> </name><name name-style="western"><surname>Alizai</surname><given-names>MH</given-names> </name></person-group><article-title>Automation bias in large language model&#x2013;assisted diagnostic reasoning among physicians trained in AI literacy &#x2014; a randomized clinical trial</article-title><source>NEJM AI</source><year>2026</year><month>04</month><day>23</day><volume>3</volume><issue>5</issue><pub-id pub-id-type="doi">10.1056/AIoa2501001</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parasuraman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Manzey</surname><given-names>DH</given-names> </name></person-group><article-title>Complacency and bias in human use of automation: an attentional integration</article-title><source>Hum Factors</source><year>2010</year><month>06</month><volume>52</volume><issue>3</issue><fpage>381</fpage><lpage>410</lpage><pub-id pub-id-type="doi">10.1177/0018720810376055</pub-id><pub-id pub-id-type="medline">21077562</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Skitka</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Mosier</surname><given-names>K</given-names> </name><name name-style="western"><surname>Burdick</surname><given-names>MD</given-names> </name></person-group><article-title>Accountability and automation bias</article-title><source>Int J Hum Comput Stud</source><year>2000</year><month>04</month><volume>52</volume><issue>4</issue><fpage>701</fpage><lpage>717</lpage><pub-id pub-id-type="doi">10.1006/ijhc.1999.0349</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Uptated EHEALTH CONSORT checklist.</p><media xlink:href="formative_v10i1e104579_app1.pdf" xlink:title="PDF File, 8534 KB"/></supplementary-material></app-group></back></article>