<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JFR</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id>
      <journal-title>JMIR Formative Research</journal-title>
      <issn pub-type="epub">2561-326X</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v10i1e103419</article-id>
      <article-id pub-id-type="pmid">42615486</article-id>
      <article-id pub-id-type="doi">10.2196/103419</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Patient-Facing AI System for Symptom Guidance Using Simulated Encounters and Physician Review: Analytical Validation Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Mavragani</surname>
            <given-names>Amaryllis</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Koch, Prof. Dr.</surname>
            <given-names>Roland</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Kopka</surname>
            <given-names>Marvin</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Patwardhan</surname>
            <given-names>Anil</given-names>
          </name>
          <degrees>BS, PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Verily Life Sciences LLC (United States)</institution>
            <addr-line>2999 Olympus Blvd, Ste 1000</addr-line>
            <addr-line>Dallas, TX, 75019</addr-line>
            <country>United States</country>
            <phone>1 4155197666</phone>
            <email>patwardh@verily.health</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-2816-7142</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Verma</surname>
            <given-names>Nishant</given-names>
          </name>
          <degrees>BTECH, MS, PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-4126-0648</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Barman</surname>
            <given-names>Poulami</given-names>
          </name>
          <degrees>BS, MS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-4604-9868</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Mao</surname>
            <given-names>Bingyu</given-names>
          </name>
          <degrees>BS, MS, PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-5428-8079</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Li</surname>
            <given-names>Nan</given-names>
          </name>
          <degrees>BS, MS, PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0009-9802-2630</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Varghese</surname>
            <given-names>Paul</given-names>
          </name>
          <degrees>BS, MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0005-2941-2310</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Peri</surname>
            <given-names>Vidya</given-names>
          </name>
          <degrees>BBA</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-0366-0866</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author">
          <name name-style="western">
            <surname>Lehmann</surname>
            <given-names>Lisa Soleymani</given-names>
          </name>
          <degrees>BA, MS, MD, PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-8779-1244</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Verily Life Sciences LLC (United States)</institution>
        <addr-line>Dallas, TX</addr-line>
        <country>United States</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Anil Patwardhan <email>patwardh@verily.health</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>19</day>
        <month>8</month>
        <year>2026</year>
      </pub-date>
      <volume>10</volume>
      <elocation-id>e103419</elocation-id>
      <history>
        <date date-type="received">
          <day>4</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>24</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>22</day>
          <month>7</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>28</day>
          <month>7</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Anil Patwardhan, Nishant Verma, Poulami Barman, Bingyu Mao, Nan Li, Paul Varghese, Vidya Peri, Lisa Soleymani Lehmann. Originally published in JMIR Formative Research (https://formative.jmir.org), 19.08.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on https://formative.jmir.org, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://formative.jmir.org/2026/1/e103419" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Rapid advances in large language models (LLMs) have expanded interest in patient-facing health applications that support symptom assessment and care-seeking decisions. Although AI-enabled symptom-guidance tools could improve patient navigation and recognition of clinically serious conditions, inappropriate recommendations may result in missed needed care or unnecessary escalation. Structured predeployment evaluation is therefore needed before prospective clinical use.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This formative predeployment study aimed to analytically validate a locked configuration of the Personal Health Assistant (PHA; Verily Health), an LLM-enabled patient-facing symptom-guidance tool, under controlled conditions using simulated patient encounters and expert physician review.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We conducted a prospective predeployment analytical validation study using synthetic patient data. The analysis set included 772 synthetic cases generated using published telephone-triage protocols and persona-based, LLM-assisted case generation. Each case included a clinical vignette, structured medical history information, and a simulated multiturn patient conversation. All cases were reviewed by a panel of 5 independent physicians without access to PHA outputs in order to provide level-of-care recommendations. Case-level ground truth was derived from aggregated physician ratings using prespecified plurality and tie-handling rules. Urgent undertriage, nonurgent undertriage, and overtriage were evaluated against prespecified performance criteria. Secondary analyses assessed the clinical appropriateness of PHA-generated possible conditions and escalation, self-care, and laboratory-testing recommendations.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>The final analysis set contained 772 cases, including 2 with unevaluable ground truth that were excluded from primary Endpoint denominators. Urgent undertriage was identified in 28 of 406 cases (6.9%, 95% CI 4.6%-9.8%), nonurgent undertriage in 56 of 305 (18.4%, 95% CI 14.2%-23.2%), and overtriage in 40 of 364 (11.0%, 95% CI 8.0%-14.7%). Overtriage met the prespecified criterion of less than 30%, whereas urgent and nonurgent undertriage did not meet their respective criteria of less than 5% and less than 15%, respectively. Secondary findings were generally supportive of the clinical appropriateness of the additional PHA outputs.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>PHA met the prespecified overtriage criterion but did not meet the urgent or nonurgent undertriage criteria. The findings identified undertriage as the principal performance gap requiring further product refinement. Because the study used synthetic cases, simulated conversations, and enriched urgency categories under controlled conditions, the results may not reflect performance in real-world populations or deployment settings. Prospective evaluation with actual users in the intended-use population is needed before broader clinical use.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>symptom guidance</kwd>
        <kwd>symptom checker</kwd>
        <kwd>large language model</kwd>
        <kwd>triage</kwd>
        <kwd>analytical validation</kwd>
        <kwd>digital health</kwd>
        <kwd>remote care</kwd>
        <kwd>undertriage</kwd>
        <kwd>overtriage</kwd>
        <kwd>simulation</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <sec>
        <title>Background</title>
        <p>Rapid advances in large language models (LLMs) have created new opportunities for health care applications that depend on reasoning, synthesis, and decision support. One emerging application is patient-facing symptom guidance, in which digital assistants attempt to guide patients toward an appropriate level of care based on information provided directly by the patient about their symptoms. Symptom checkers and related digital tools have traditionally relied on rules-based systems [<xref ref-type="bibr" rid="ref1">1</xref>], but newer systems increasingly incorporate AI components [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>] to support more flexible interaction and richer contextual interpretation. This shift to integrating generative AI may allow patient-facing tools to move beyond branching logic toward more adaptive exchanges that better incorporate symptom context and, when available, relevant medical history.</p>
        <p>A reliable AI-enabled symptom-guidance tool could help patients interpret symptoms and make more appropriate care-seeking decisions, improve navigation to the appropriate level of care, reduce unnecessary usage for low-acuity concerns, and enhance recognition of urgent conditions. At the same time, such a tool may also introduce clinically important errors if it provides inconsistent recommendations, fails to recognize cases that warrant urgent or nonurgent follow-up care, or escalates low-acuity cases unnecessarily. A central challenge for these patient-facing guidance tools is minimizing errors related to missed needed care while avoiding unnecessary escalation.</p>
        <p>Prior studies of digital symptom assessment show variable and often imperfect triage performance. Evaluations of symptom-assessment apps reveal substantial heterogeneity, with performance commonly reflecting a tradeoff between risk-averse escalation and failure to identify time-sensitive conditions [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. Separate undertriage and overtriage Endpoints distinguish the direction and potential consequences of triage errors; however, these Endpoints have been reported less consistently than overall triage accuracy. More recent evaluations of LLM-based symptom checkers suggest that LLM triage performance may approach that of health care professionals in some settings [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. However, a structured stress test of ChatGPT Health (OpenAI) found that failures were concentrated at the clinical extremes, including undertriage of emergency conditions, while other work has also reported variability in care-seeking advice and limited real-world evaluation [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>].</p>
        <p>The evaluation literature is beginning to move beyond static benchmark cases toward more realistic conversational assessment. For example, SymptomAI used naturalistic symptom conversations from lay users and showed that active symptom interviewing improved differential-diagnosis accuracy compared with user-guided chatbot-style interactions [<xref ref-type="bibr" rid="ref11">11</xref>]. This finding underscores that patient-facing AI symptom tools should be evaluated as interactive systems, because performance depends not only on model reasoning but also on whether the system elicits clinically relevant information through the conversation. However, differential-diagnosis accuracy and level-of-care accuracy raise different questions. For patient-facing symptom-guidance products, the most immediate safety-critical question is whether the system recommends the appropriate level of care. The Symptom Checker Accuracy Reporting Framework (SCARF) was recently proposed to improve transparency and consistency in studies of symptom-checker accuracy by providing guidance on case selection, reference-standard assignment, evaluation design, and outcome reporting [<xref ref-type="bibr" rid="ref12">12</xref>]. Although SCARF provides an important reporting foundation, product-specific evaluations of interactive systems must also determine how to represent intended-use conversations, use a sufficiently large ground-truth panel to account for interphysician variability, define clinically meaningful undertriage and overtriage Endpoints, and prespecify performance criteria and sample size requirements.</p>
      </sec>
      <sec>
        <title>Objective</title>
        <p>This analytical validation (AV) study was designed to address a methodological gap in the evaluation of patient-facing AI tools. We applied a structured predeployment validation approach to a defined investigational product. The study used prespecified Endpoints anchored on clinically important undertriage and overtriage rates, clinically structured cases rooted in established triage protocols, simulated interactive encounters rather than static vignettes alone, and blinded expert physician review with prespecified ground-truth derivation rules. Because the investigational tool is designed to use historical medical information when available, the synthetic cases also incorporated medical history context.</p>
        <p>The primary aim was to determine whether a locked configuration of Personal Health Assistant (PHA) met the prespecified performance criteria for urgent undertriage, nonurgent undertriage, and overtriage. Secondary aims were to assess the clinical appropriateness of PHA’s possible-condition, escalation-guidance, self-care-guidance, and laboratory-testing outputs. This study was intended as an early evidentiary step to determine whether the tool performed acceptably on a large set of known test cases under controlled conditions before proceeding to prospective clinical studies. More broadly, the study illustrates an approach to predeployment evaluation of patient-facing AI systems for symptom guidance that emphasizes clinically relevant error Endpoints, intended-use simulation, and blinded expert review alongside overall performance.</p>
      </sec>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>The PHA and Evaluated Configuration</title>
        <p>The PHA is a patient-facing LLM-enabled symptom-guidance tool that supports conversational assessment of symptoms. It operates as a multiturn agent that combines conversational symptom collection, screening for ineligibility or emergency presentations, retrieval and summarization of relevant medical records from a Health Information Exchange (HIE), iterative refinement through follow-up questioning, and generation of patient-facing outputs. PHA generates a level-of-care recommendation and, for cases assessed as appropriate for nonurgent follow-up care or self-care, also generates possible conditions, escalation guidance, self-care guidance, and recommendations for laboratory testing or other follow-up evaluation. It is intended for adults aged 18 years and older experiencing health-related symptoms in a remote setting and is accessible through a mobile app (<xref rid="figure1" ref-type="fig">Figure 1</xref>). Because its outputs may influence how patients interpret their symptoms and whether and when they seek care, careful evaluation is needed before it can be responsibly deployed for general clinical use.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Overview of the Personal Health Assistant (PHA). The PHA is a patient-facing symptom guidance tool that integrates patient-reported symptoms through an interactive conversation and relevant health information exchange (HIE) data. Based on these inputs, PHA generates a level-of-care recommendation (ie, seek immediate medical care, seek nonurgent follow-up care, or pursue self-care and monitoring), and a list of 5 possible conditions. When applicable, it outputs self-care instructions, escalation instructions, and lab/testing recommendations.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e103419_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>A prespecified version of PHA was locked before execution on the AV test cases, and all results refer to that fixed evaluated configuration. For the version evaluated in this study, PHA was based on the OpenAI GTP-5.4-mini LLM (2026-03-17 model release date). In lieu of the retrieval and summarization of medical records from HIE, this version of PHA required a minimum number of direct patient input fields that included age, sex at birth, social history, immunization summary, known allergies, recent encounter information, chronic conditions, recent laboratory data, and recent diagnoses. These evaluated-version constraints are important for interpretation of the AV results.</p>
      </sec>
      <sec>
        <title>Study Design</title>
        <p>This study used synthetic patient data to evaluate a locked configuration of PHA under controlled predeployment conditions. Each case combined synthetic patient information, synthetic medical record information, and a simulated patient conversation. PHA outputs and physician-derived reference assessments were collected prospectively for the validation cases. PHA was run on each case to generate patient-facing symptom-guidance outputs (<xref rid="figure2" ref-type="fig">Figure 2</xref>), and independent physician reviewers evaluated the same cases using a prespecified grading rubric (<xref rid="figure3" ref-type="fig">Figure 3</xref>). The physician review process included blinded and unblinded components to support independent ground-truth derivation for primary Endpoints and appropriateness assessment of selected PHA outputs for secondary Endpoints.</p>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Study workflow for obtaining the Personal Health Assistant (PHA) outputs from synthetic data. A synthetic vignette is used to prompt a patient-simulator, which interacts with the PHA to generate a simulated multi-turn conversation. In combination with medical history information, this conversation enables PHA to generate its outputs.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e103419_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Blinded and unblinded physician review workflow. In the blinded review, 5 physicians independently reviewed only the case materials to provide a level-of-care recommendation and listed up to 5 possible conditions; ground truth for level-of-care was derived by plurality vote with prespecified tie rules. In unblinded review, physicians reviewed the same case materials together with selected PHA outputs and rated agreement with PHA’s outputs of possible conditions and recommendations. GT: ground truth; PHA: Personal Health Assistant.</p>
          </caption>
          <graphic xlink:href="formative_v10i1e103419_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>In the blinded review, all cases were reviewed by a panel of 5 independent physicians, resulting in at least 3 physician ratings available per case. Physicians assigned level of care and listed up to 5 possible conditions using the case materials only; ground truth for level of care was derived by plurality vote with prespecified tie rules. In unblinded review, physicians reviewed the same case materials together with selected PHA outputs and rated agreement with PHA’s outputs of possible conditions and recommendations.</p>
      </sec>
      <sec>
        <title>Synthetic Test Case Development and Quality Control</title>
        <p>The AV dataset consisted of synthetic test cases representing distinct patient encounters. Each test case contained two related but distinct components: (1) a simulated vignette used to instantiate the patient simulator and (2) a set of structured synthetic medical history fields made directly available to PHA (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The vignette summarized the encounter in natural language and included sex at birth, age, social history, family history, medical history, review of systems, history of present illness, current medications, immunizations, allergies, and the chief symptom expressed in patient-facing language. The structured fields made directly available to PHA at the start of the interaction included sex at birth, age, social history, immunization summary, known allergies, chronic conditions, a patient-encounter summary from the prior 12 months, recent laboratory data, and recent diagnoses. PHA then received additional information during its multiturn interaction with the patient simulator.</p>
        <p>Briggs’ Telephone Triage Protocols for Nurses, sixth edition [<xref ref-type="bibr" rid="ref13">13</xref>], was used to assign putative triage urgency to each case and to guide case selection, ensuring adequate numbers of urgent, nonurgent, and self-care cases for evaluation of the primary Endpoints. These protocol-based assignments served only as an initial proxy for urgency during dataset construction; final ground truth (GT) was established subsequently through physician review. A total of 800 synthetic cases were generated in a 400:300:100 allocation of urgent, nonurgent, and self-care categories. Putative triage urgency was not available to PHA or to physician reviewers during evaluation.</p>
        <p>Synthetic case generation followed a structured process anchored to the published triage protocols [<xref ref-type="bibr" rid="ref13">13</xref>] and a persona-based generation approach. Triage protocol logic was decomposed into reusable clinical templates corresponding to different symptom pathways and urgency categories. Patient personas were then varied systematically across social history, family history, medical history, symptom trajectory, and related clinical context to produce a large number of unique but medically coherent scenarios. Structured clinical fields were generated to create plausible longitudinal context for each encounter, and the vignette summarized the broader case narrative in natural language for use by the patient simulator. A detailed description of the case generation process, including example prompts and case structure, is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p>
        <p>Cases were created across a broad symptom set selected to support coverage of urgent, nonurgent, and self-care scenarios. Symptom selection was informed by the triage protocols and further enriched with symptoms identified by Verily Health’s clinical team as important to triage correctly, including symptoms that are often benign but may escalate depending on associated history or accompanying symptoms. The full symptom list is provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p>
        <p>An independent quality control review was performed before evaluation to confirm that the synthetic cases were suitable for testing. A total of 200 cases were randomly selected from the parent set using stratified sampling by putative triage urgency to reflect the parent-set distribution (50% urgent, 37.5% nonurgent, and 12.5% self-care). Review was performed by an independent physician distinct from the physicians later used for GT derivation. The review assessed whether vignettes could reasonably support assignment of a level-of-care category in a remote symptom-guidance setting; whether they were clear and understandable, accurately summarized the key fields, avoided explicit leakage of triage recommendation or other downstream outputs, contained no gross physiologic, anatomic, or medical inconsistencies; and whether the medical history fields were compatible with the vignette.</p>
      </sec>
      <sec>
        <title>Simulated Conversation Component</title>
        <p>Each AV case also included a simulated patient conversation. This component was included because PHA was evaluated in an interactive setting rather than on static summary information alone. Simulated conversations were generated using a patient simulator developed previously for conversational evaluation of health care AI agents [<xref ref-type="bibr" rid="ref14">14</xref>]. In that prior work, the simulator was grounded in real-world electronic health record (EHR)-derived patient vignettes and generated scalable, privacy-preserving, multiturn conversational encounters for development and evaluation of health care AI systems. This prior evaluation assessed more than 500 simulated encounters, with clinician review demonstrating high consistency between simulated conversations and the source vignettes.</p>
        <p>For this study, the simulator instantiated each case from the underlying vignette and generated natural-language responses that remained grounded in the source case while disclosing information only as elicited through the interaction. This allowed the simulator to render realistic multiturn encounters rather than simply restating the full case summary at once. A quality control review also assessed whether simulated conversations contained explicit leakage of downstream outputs. When simulated conversation transcripts contained wording that directly implied level of care, that content was manually removed to preserve the independence of physician review.</p>
      </sec>
      <sec>
        <title>Physician Review and Ground-Truth Derivation</title>
        <p>Physician reviewers were required to have US medical school training, at least 8 years of posttraining practice experience in the United States, board certification, and consistent practice in emergency medicine or internal medicine, including relevant internal medicine subspecialties. Before formal review began, all reviewers completed study-specific onboarding that included written instructions, orientation to the grading workflow, and a pilot or practice run in which their responses were reviewed for alignment with the grading instructions. Reviewers were instructed to rely on their own clinical judgment based only on the information presented in the study interface and not to use external references or tools during review.</p>
        <p>Cases were assigned to reviewers in 2 batches using a partially crossed design. In the first batch, cases were assigned to physicians randomly drawn from a pool of 11 reviewers; in the second batch, reviewers were drawn from a pool of 13. In both batches, assignment was constructed so that each case was assigned to 5 physicians, no case included a duplicate reviewer, workload was distributed approximately evenly across reviewers, and reviewer pairings were balanced so that no specific pair worked together excessively. 332 cases were later reassigned when originally assigned reviewers were unable to complete grading within the required timeframe. Reassignments were distributed across the reviewer pool to preserve workload balance and maintain the partially crossed design. Reviewers worked independently on each case using an online case-review tool [<xref ref-type="bibr" rid="ref15">15</xref>] and had no access to the responses of other reviewers.</p>
        <p>The grading workflow included blinded and unblinded components (<xref rid="figure3" ref-type="fig">Figure 3</xref>) that served different evaluation objectives. In the blinded component, all cases were reviewed by a panel of 5 independent physicians, resulting in at least 3 physician ratings available per case. Reviewers assessed the synthetic vignette, simulated patient conversation, and structured clinical fields without access to PHA outputs and independently assigned level of care and up to 5 possible conditions. The blinded level-of-care assessments were used to derive a single ground-truth label per case using plurality vote with prespecified tie rules. In the unblinded component, physicians reviewed the same case materials together with selected PHA outputs and rated their agreement with the PHA possible conditions, escalation guidance, care recommendations, and lab and testing recommendations. This structure allowed independent reference assessment for the primary Endpoint while also supporting evaluation of the clinical relevance of additional patient-facing outputs. Full reviewer instructions and workflow are provided in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p>
        <p>GT for level of care was derived from the blinded physician ratings using prespecified aggregation rules. A case required at least 3 evaluable physician ratings to be eligible for GT assignment. Cases with fewer than 3 available ratings were excluded from analysis. For eligible cases, GT was assigned by plurality vote across the urgency categories. When a unique plurality was not present, prespecified tie rules were applied. In general, ties across clinical urgency categories were resolved toward the higher urgency category. For example, in a 2:2:1 split across urgency levels, GT was assigned to the higher of the 2 tied categories. When 2 reviewers assigned a case as unevaluable and 2 assigned the same level-of-care category, the case was assigned as unevaluable.</p>
      </sec>
      <sec>
        <title>Endpoints</title>
        <p>PHA generated a case-level level-of-care recommendation for each encounter in 1 of 3 categories: seek immediate medical care, indicating that prompt evaluation in an emergency or otherwise urgent care setting was recommended; seek nonurgent follow-up care, indicating that follow-up with a clinician was recommended but immediate emergency evaluation was not; or pursue self-care and monitoring, indicating that home management and observation were recommended without immediate clinician evaluation. These assignments were compared with physician-derived GT classifications of urgent, nonurgent, or self-care. Primary performance was summarized using 3 population-level mistriage measures derived from that comparison: urgent undertriage, defined as the proportion of urgent GT cases assigned a lower level of care by PHA; nonurgent undertriage, defined as the proportion of nonurgent GT cases assigned to pursue self-care and monitoring by PHA; and overtriage, defined as the proportion of nonurgent or self-care GT cases assigned a higher level of care by PHA.</p>
        <p>Secondary Endpoints evaluated PHA outputs beyond the level-of-care recommendation itself. These outputs were intended to provide additional clinical context and guidance, including possible conditions, which summarized conditions that might explain the presentation; escalation guidance, which described when medical attention should be sought if symptoms worsened or failed to improve; self-care guidance, which provided supportive home-management advice; and lab and testing recommendations, which suggested possible follow-up tests or evaluations. Whereas the primary Endpoints assessed whether PHA assigned the appropriate urgency category, the secondary Endpoints assessed whether the supporting explanatory and management content generated by PHA was clinically appropriate for the case. Possible conditions were evaluated by comparison with physician-derived condition lists and by physician agreement assessment. Escalation, self-care, and lab and testing recommendations were evaluated by physician agreement regarding their appropriateness for the case.</p>
      </sec>
      <sec>
        <title>Prespecified Performance Thresholds and Clinical Rationale</title>
        <p>Prespecified performance thresholds were defined for urgent undertriage, nonurgent undertriage, and overtriage before analysis. Because no single authoritative consensus defines acceptable mistriage rates for adult, nontrauma, remote symptom-guidance systems, thresholds were selected by triangulating adjacent clinical guidance, published observational evidence, and the practical constraints of remote assessment without physical examination. Thresholds were stratified by urgency level because the clinical consequences of misclassification differ across levels of acuity.</p>
        <p>A threshold of &#60;5% was prespecified for urgent undertriage as a conservative, safety-focused benchmark. This threshold was informed by remote-care evidence reviews citing American College of Surgeons Committee on Trauma (ACS-COT) guidance as a proxy reference where more specific adult remote-triage standards are lacking [<xref ref-type="bibr" rid="ref16">16</xref>], by adjacent studies reporting undertriage rates [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>], and by the practical recognition that zero urgent undertriage is unrealistic in exam-free remote assessment. A more permissive threshold of &#60;15% was prespecified for nonurgent undertriage, reflecting the lower immediate safety risk of these errors and the wider variability reported in remote and digital triage studies. An overall threshold of &#60;30% was prespecified for overtriage to balance the competing goal of limiting unnecessary escalation while remaining within adjacent benchmark guidance and below many published clinician-led and digital-triage comparators [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref19">19</xref>].</p>
      </sec>
      <sec>
        <title>Statistical Analysis</title>
        <p>The unit of analysis was a single synthetic patient-encounter, defined as a case comprising the vignette, simulated conversation, and structured patient information. The analysis dataset included cases with GT assigned from at least 3 physician ratings, including cases assigned an unevaluable GT category. All cases in the analysis dataset received a PHA level-of-care output; cases with unevaluable GT were retained in descriptive summaries but excluded from primary Endpoint denominators.</p>
        <p>Primary performance was summarized using 3 population-level mistriage measures: urgent undertriage, nonurgent undertriage, and overtriage. Urgent undertriage was calculated among cases with urgent GT as the proportion assigned a lower level of care by PHA. Nonurgent undertriage was calculated among cases with nonurgent GT as the proportion assigned to pursue self-care and monitoring by PHA. Overtriage was calculated among cases with nonurgent or self-care GT as the proportion assigned a higher level of care by PHA. For each coprimary Endpoint, point estimates and 95% exact Clopper-Pearson CIs were calculated. Performance was evaluated against prespecified thresholds using the upper bound of the 2-sided 95% CI, equivalent to 1-sided exact binomial testing at α=.025.</p>
        <p>Supportive analyses examined the robustness of the primary Endpoint results to alternative GT tie rules in which ties were resolved toward a lower level-of-care assignment and to subsets of cases with stronger reviewer agreement, including cases where (1) there were no-ties among physicians; (2) full-consensus subsets, in which all physicians agreed; and (3) cases where all physician reviewers provided an evaluable level-of-care assessment (ie, no unevaluable votes).</p>
        <p>A modified analysis population was used for secondary Endpoints related to possible conditions and recommendation outputs. This population excluded cases in which PHA recommended “seek immediate medical care” because these secondary outputs were only generated for cases classified by PHA as appropriate for nonurgent follow-up care or self-care.</p>
        <p>For possible conditions, free-text PHA condition terms and physician-derived condition terms were matched using Jaro-Winkler similarity. Four overlap-based metrics were then calculated for each reviewer-case pair: hit rate, indicating whether PHA captured at least 1 physician-listed condition; recall, the proportion of physician-listed conditions captured by PHA; precision, the proportion of PHA-listed conditions captured by physician-listed conditions; and <italic>F</italic><sub>1</sub>-score, the harmonic mean of recall and precision. Agreement with PHA’s possible conditions output was also summarized as the proportion of cases in which a majority of physician reviewers judged that the PHA list included at least 1 clinically appropriate condition.</p>
        <p>For escalation guidance, self-care guidance, and lab and testing recommendations, physician agreement with PHA outputs was rated on a 5-point Likert scale from strongly disagree to strongly agree, reflecting whether the recommendation was appropriate for the case. Physician agreement with recommendation outputs was summarized across individual grader-case ratings using mean Likert scores with 95% bootstrap CIs, together with the distribution of ratings across Likert categories.</p>
        <p>The study design targeted approximately 400 urgent GT cases, 300 nonurgent GT cases, and 100 self-care cases, for a total of 800 synthetic cases. Based on simulation using single-sample proportion testing assumptions, this design was expected to provide approximately 71% overall power across the 3 coprimary Endpoints under the planning assumption of 5 physician reviewers per case. Additional details regarding sample size justification, supportive analyses, and secondary Endpoint scoring are provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>This study used exclusively synthetic patient cases, synthetic medical history information, and simulated patient conversations. It did not involve living individuals, identifiable private information, or biospecimens. Before study execution, the project was reviewed by the Verily Human Subjects Research Committee, which determined that the activity did not constitute human subjects research requiring institutional review board approval under 45 CFR 46. Because no human research participants or identifiable human data were involved, informed consent was not required.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Analysis Sample</title>
        <p>A total of 799 cases entered the analytical pipeline from the original 800 synthetically constructed cases, reflecting removal of 1 case because it involved a patient younger than 18 years. During final analysis, 26 cases were identified as duplicates based on identical or near-identical patient summaries representing essentially the same clinical scenario, and 1 additional case was excluded because it involved a patient younger than 18 years. The final analysis dataset therefore contained 772 cases (799 − 26 − 1). There was no loss of cases due to missing PHA outputs, and every case had at least 3 physician ratings available for GT derivation. Of the 772 cases, 770 had an urgent, nonurgent, or self-care GT classification and contributed to at least 1 primary Endpoint denominator; 2 had an unevaluable GT classification and were retained in descriptive summaries but excluded from the primary Endpoint denominators. Mean age was 48.9 (SD 17.6) years, median age was 48 (range 18-85) years, and 410/772 (53.1%) cases were female.</p>
        <p>Among the 772 cases, 377 had a PHA level-of-care recommendation other than “seek immediate medical care” and therefore generated secondary outputs for possible conditions, escalation guidance, self-care guidance, and lab and testing recommendations. These 377 cases comprised the modified analysis population used for the secondary Endpoints.</p>
        <p>To summarize clinical coverage of the AV dataset, <xref ref-type="table" rid="table1">Table 1</xref> groups the cases by presenting-symptom category based on the chief complaint. Categories were assigned from free-text chief symptom descriptions for the purpose of summarizing clinical coverage. For each category, the number of urgent, nonurgent, and self-care cases are shown as determined by physician-derived GT assessments. The full dataset showing symptoms among all cases is available in the Supplementary Materials.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Presenting symptom categories of Personal Health Assistant cases and physician-derived ground truth.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="210"/>
            <col width="260"/>
            <col width="80"/>
            <col width="130"/>
            <col width="100"/>
            <col width="140"/>
            <col width="80"/>
            <thead>
              <tr valign="top">
                <td>Presenting symptom category</td>
                <td>Representative chief complaints</td>
                <td>Urgent, n (%)</td>
                <td>Nonurgent, n (%)</td>
                <td>Self-care, n (%)</td>
                <td>Unevaluable, n (%)</td>
                <td>Total, n (%)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Neurologic</td>
                <td>Headache, dizziness, weakness, focal neurologic symptoms</td>
                <td>142 (18.4)</td>
                <td>41 (5.3)</td>
                <td>5 (0.6)</td>
                <td>0 (0.0)</td>
                <td>188 (24.4)</td>
              </tr>
              <tr valign="top">
                <td>Abdominal, gastrointestinal</td>
                <td>Abdominal pain, nausea, vomiting, bowel symptoms</td>
                <td>60 (7.8)</td>
                <td>40 (5.2)</td>
                <td>6 (0.8)</td>
                <td>0 (0.0)</td>
                <td>106 (13.7)</td>
              </tr>
              <tr valign="top">
                <td>Musculoskeletal/injury</td>
                <td>Back or joint pain, limb pain, swelling, injury</td>
                <td>42 (5.4)</td>
                <td>35 (4.5)</td>
                <td>15 (1.9)</td>
                <td>0 (0.0)</td>
                <td>92 (11.9)</td>
              </tr>
              <tr valign="top">
                <td>Cardiopulmonary</td>
                <td>Chest pain, dyspnea, cough, wheeze, palpitations</td>
                <td>67 (8.7)</td>
                <td>20 (2.6)</td>
                <td>3 (0.4)</td>
                <td>0 (0.0)</td>
                <td>90 (11.7)</td>
              </tr>
              <tr valign="top">
                <td>Genitourinary, reproductive, breast</td>
                <td>Dysuria, flank pain, urinary or pelvic symptoms, breast symptoms</td>
                <td>28 (3.6)</td>
                <td>57 (7.4)</td>
                <td>2 (0.3)</td>
                <td>1 (0.1)</td>
                <td>88 (11.4)</td>
              </tr>
              <tr valign="top">
                <td>Mental, behavioral</td>
                <td>Anxiety, panic-like symptoms, behavioral complaints</td>
                <td>8 (1.0)</td>
                <td>28 (3.6)</td>
                <td>5 (0.6)</td>
                <td>0 (0.0)</td>
                <td>41 (5.3)</td>
              </tr>
              <tr valign="top">
                <td>Eye, skin, soft tissue</td>
                <td>Eye irritation, rash, pruritus, skin complaints</td>
                <td>16 (2.1)</td>
                <td>16 (2.1)</td>
                <td>8 (1.0)</td>
                <td>0 (0.0)</td>
                <td>40 (5.2)</td>
              </tr>
              <tr valign="top">
                <td>Constitutional, general</td>
                <td>Fever, chills, fatigue, generalized symptoms</td>
                <td>7 (0.9)</td>
                <td>15 (1.9)</td>
                <td>4 (0.5)</td>
                <td>0 (0.0)</td>
                <td>26 (3.4)</td>
              </tr>
              <tr valign="top">
                <td>ENT<sup>a</sup>, oral</td>
                <td>Sore throat, ear pain, oral or dental complaints</td>
                <td>3 (0.4)</td>
                <td>13 (1.7)</td>
                <td>3 (0.4)</td>
                <td>0 (0.0)</td>
                <td>19 (2.5)</td>
              </tr>
              <tr valign="top">
                <td>Endocrine, metabolic</td>
                <td>Thirst, polyuria, heat intolerance, glucose-related symptoms</td>
                <td>3 (0.4)</td>
                <td>10 (1.3)</td>
                <td>3 (0.4)</td>
                <td>0 (0.0)</td>
                <td>16 (2.1)</td>
              </tr>
              <tr valign="top">
                <td>Other</td>
                <td>Mixed vascular, swallowing, and miscellaneous complaints</td>
                <td>30 (3.9)</td>
                <td>30 (3.9)</td>
                <td>5 (0.6)</td>
                <td>1 (0.1)</td>
                <td>66 (8.5)</td>
              </tr>
              <tr valign="top">
                <td>Total</td>
                <td>—<sup>b</sup></td>
                <td>406 (52.6)</td>
                <td>305 (39.5)</td>
                <td>59 (7.6)</td>
                <td>2 (0.3)</td>
                <td>772 (100.0)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>ENT: ear, nose, and throat.</p>
            </fn>
            <fn id="table1fn2">
              <p><sup>b</sup>Not applicable.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Primary Endpoints</title>
        <p>The confusion matrix in <xref ref-type="table" rid="table2">Table 2</xref> summarizes the distribution of triage assignments by physician-derived GT and by PHA, where rows represent physician-derived ground-truth categories and columns represent PHA level-of-care recommendations. All cases received at least 3 physician ratings; however, 2 cases were excluded from the primary Endpoint denominators because their GT classification was unevaluable. Of the 770 cases with evaluable GT, 406 were classified as urgent, 305 as nonurgent, and 59 as self-care. Across all 772 cases, PHA classified 395 as “seek immediate medical care”, 284 as “seek nonurgent follow-up care”, and 93 as “pursue self-care and monitoring”, with no unevaluable outputs. The 3 coprimary Endpoint results derived from this confusion matrix are summarized in <xref ref-type="table" rid="table3">Table 3</xref>. Urgent undertriage was identified in 28 of 406 cases (6.9%, 95% CI 4.6%-9.8%), nonurgent undertriage in 56 of 305 (18.4%, 95% CI 14.2%-23.2%), and overtriage in 40 of 364 (11.0%, 95% CI 8.0%-14.7%). A successful Endpoint required the upper bound of the 95% CI to fall below the prespecified clinical threshold. Relative to the prespecified performance thresholds, the overtriage Endpoint met its margin of 30%, whereas both urgent undertriage and nonurgent undertriage exceeded their respective thresholds of 5% and 15%.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Confusion matrix of physician-derived ground truth versus Personal Health Assistant level-of-care recommendations in the analysis set.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="230"/>
            <col width="250"/>
            <col width="250"/>
            <col width="270"/>
            <thead>
              <tr valign="bottom">
                <td>Physician-derived ground truth</td>
                <td>PHA<sup>a</sup>: seek immediate medical care, n</td>
                <td>PHA: seek nonurgent follow-up care, n</td>
                <td>PHA: pursue self-care and monitoring, n</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Urgent</td>
                <td>378</td>
                <td>27</td>
                <td>1</td>
              </tr>
              <tr valign="top">
                <td>Nonurgent</td>
                <td>17</td>
                <td>232</td>
                <td>56</td>
              </tr>
              <tr valign="top">
                <td>Self-care</td>
                <td>0</td>
                <td>23</td>
                <td>36</td>
              </tr>
              <tr valign="top">
                <td>Unevaluable</td>
                <td>0</td>
                <td>2</td>
                <td>0</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>PHA: Personal Health Assistant.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Primary Endpoint results for Personal Health Assistant level-of-care recommendations in the analysis set.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="190"/>
            <col width="180"/>
            <col width="170"/>
            <col width="170"/>
            <col width="170"/>
            <col width="120"/>
            <thead>
              <tr valign="bottom">
                <td>Primary Endpoint measure</td>
                <td>PHA<sup>a</sup> mistriage events, n</td>
                <td>Total cases, n</td>
                <td>Rate of mistriage (95% CI)</td>
                <td>Prespecified clinical threshold</td>
                <td><italic>P</italic> value</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Urgent undertriage<sup>b</sup></td>
                <td>28</td>
                <td>406</td>
                <td>0.07 (0.5 to 0.10)</td>
                <td>0.05</td>
                <td>0.752</td>
              </tr>
              <tr valign="top">
                <td>Nonurgent undertriage<sup>c</sup></td>
                <td>56</td>
                <td>305</td>
                <td>0.18 (0.14 to 0.23)</td>
                <td>0.15</td>
                <td>0.941</td>
              </tr>
              <tr valign="top">
                <td>Overtriage<sup>d</sup></td>
                <td>40</td>
                <td>364</td>
                <td>0.11 (0.08 to 0.15)</td>
                <td>0.30</td>
                <td>&#60;.001</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>PHA: Personal Health Assistant.</p>
            </fn>
            <fn id="table3fn2">
              <p><sup>b</sup>Ground truth = urgent; PHA ∈ {seek nonurgent follow-up care; pursue self-care and monitoring}.</p>
            </fn>
            <fn id="table3fn3">
              <p><sup>c</sup>Ground truth = nonurgent; PHA = pursue self-care and monitoring.</p>
            </fn>
            <fn id="table3fn4">
              <p><sup>d</sup>Ground truth ∈ {nonurgent follow-up, self-care}; PHA more urgent than ground truth.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>Most urgent ground-truth cases were correctly recognized by PHA, but a minority were assigned lower-acuity recommendations. Of the 28 urgent undertriaged cases, 27 were assigned to seek nonurgent follow-up care and 1 was assigned to pursue self-care and monitoring. Among the 305 nonurgent ground-truth cases, the dominant error mode was downgrading to self-care: 56 cases (18.4%) were assigned to pursue self-care and monitoring, whereas 17 cases (5.6%) were escalated to seek immediate medical care. Among the 59 self-care ground-truth cases, 23 (39.0%) were assigned to seek nonurgent follow-up care, and no self-care cases were assigned to seek immediate medical care.</p>
      </sec>
      <sec>
        <title>Sensitivity and Subset Analyses</title>
        <p>Because GT in this study reflected aggregated physician judgment rather than a single deterministic reference standard, supportive analyses were performed to assess robustness to alternative GT tie handling and to different degrees of reviewer agreement. This design was intentional: plurality-based GT retained clinically ambiguous and borderline presentations in the evaluation, whereas a single-reviewer reference standard would have been less informative and would not have captured interclinician disagreement.</p>
        <p>In the primary analysis, ties among level-of-care assignments were resolved toward the higher urgency category, whereas ties involving an unevaluable designation were assigned as unevaluable overall. <xref ref-type="table" rid="table4">Table 4</xref> summarizes a sensitivity analysis using the opposing tie rule, in which ties among level-of-care assignments were resolved toward the lower urgency category and ties involving unevaluable ratings were assigned to the level-of-care category with the most votes. This alternative rule produced only small shifts in the Endpoint estimates and did not change the overall conclusion. Under the alternative tie rule, urgent undertriage was identified in 22 of 393 cases (5.6%, 95% CI 3.5%-8.4%), nonurgent undertriage in 56 of 310 (18.1%, 95% CI 13.9%-22.8%), and overtriage in 52 of 377 (13.8%, 95% CI 10.5%-17.7%). As in the primary analysis, the urgent and nonurgent undertriage Endpoints did not meet their prespecified thresholds, whereas the overtriage Endpoint continued to meet its margin.</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Sensitivity analysis of primary Endpoints using alternative ground-truth tie rules.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="140"/>
            <col width="150"/>
            <col width="130"/>
            <col width="240"/>
            <col width="210"/>
            <col width="130"/>
            <thead>
              <tr valign="top">
                <td>Primary Endpoint measure</td>
                <td>Mistriage events, n</td>
                <td>Total cases, n</td>
                <td>Rate of mistriage (95% CI)</td>
                <td>Prespecified clinical threshold</td>
                <td><italic>P</italic> value</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Urgent undertriage</td>
                <td>22</td>
                <td>393</td>
                <td>0.06 (0.04-0.08)</td>
                <td>0.05</td>
                <td>.75</td>
              </tr>
              <tr valign="top">
                <td>Nonurgent undertriage</td>
                <td>56</td>
                <td>310</td>
                <td>0.18 (0.14-0.23)</td>
                <td>0.15</td>
                <td>.94</td>
              </tr>
              <tr valign="top">
                <td>Overtriage</td>
                <td>52</td>
                <td>377</td>
                <td>0.14 (0.11-0.18)</td>
                <td>0.30</td>
                <td>&#60;.01</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p><xref ref-type="table" rid="table5">Table 5</xref> summarizes the subset analyses according to reviewer agreement. The plurality-with-no-tie subset included cases in which the GT category had a unique plurality of reviewer votes, with no tie across level-of-care categories. The all-5-evaluable subset included cases in which all 5 reviewers provided evaluable level-of-care assignments (ie, no reviewer assigned the case as “unevaluable”). The full-consensus subset included only cases in which all 5 reviewers agreed on the level-of-care assignment. In the plurality-with-no-tie subset, urgent undertriage was identified in 22 of 393 cases (5.6%), nonurgent undertriage in 56 of 301 (18.6%), and overtriage in 40 of 360 (11.1%). In the all-5-evaluable subset, corresponding rates were 26 of 378 (6.9%), 54 of 279 (19.4%), and 36 of 333 (10.8%). These estimates were closely aligned with the primary analysis. In the full-consensus subset, mistriage rates were lower: urgent undertriage was identified in 3 of 312 cases (1.0%, 95% CI 0.2%-2.8%), nonurgent undertriage in 4 of 92 (4.3%, 95% CI 1.2%-10.8%), and overtriage in 1 of 97 (1.0%, 95% CI 0.0%-5.6%).</p>
        <table-wrap position="float" id="table5">
          <label>Table 5</label>
          <caption>
            <p>Analysis of under- and overtriage rates in the subset of cases.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="320"/>
            <col width="0"/>
            <col width="170"/>
            <col width="0"/>
            <col width="170"/>
            <col width="0"/>
            <col width="310"/>
            <thead>
              <tr valign="bottom">
                <td colspan="3">Subsets<sup>a</sup> and primary Endpoint measure</td>
                <td colspan="2">Mistriage events, n</td>
                <td colspan="2">Total cases, n</td>
                <td>Rate of mistriage (95% CI)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="8">Plurality (no tie)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Urgent undertriage</td>
                <td colspan="2">22</td>
                <td colspan="2">393</td>
                <td colspan="2">0.06 (0.04-0.08)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Nonurgent undertriage</td>
                <td colspan="2">56</td>
                <td colspan="2">301</td>
                <td colspan="2">0.19 (0.14-0.24)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Overtriage</td>
                <td colspan="2">40</td>
                <td colspan="2">360</td>
                <td colspan="2">0.11 (0.08-0.15)</td>
              </tr>
              <tr valign="top">
                <td colspan="8">All cases with 5 evaluable grades</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Urgent undertriage</td>
                <td colspan="2">26</td>
                <td colspan="2">378</td>
                <td colspan="2">0.07 (0.05-0.10)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Nonurgent undertriage</td>
                <td colspan="2">54</td>
                <td colspan="2">279</td>
                <td colspan="2">0.19 (0.15-0.25)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Overtriage</td>
                <td colspan="2">36</td>
                <td colspan="2">333</td>
                <td colspan="2">0.11 (0.08-0.15)</td>
              </tr>
              <tr valign="top">
                <td colspan="8">Consensus (all reviewers agree)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Urgent undertriage</td>
                <td colspan="2">3</td>
                <td colspan="2">312</td>
                <td colspan="2">0.01 (0.00-0.03)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Nonurgent undertriage</td>
                <td colspan="2">4</td>
                <td colspan="2">92</td>
                <td colspan="2">0.04 (0.01-0.11)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Overtriage</td>
                <td colspan="2">1</td>
                <td colspan="2">97</td>
                <td colspan="2">0.01 (0.00-0.06)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table5fn1">
              <p><sup>a</sup>Subset of cases: (1) clear plurality with no tie, (2) all cases with 5 reviewers and no label as “unevaluable,” and (3) cases in which all reviewers agree in the level-of-care assignment (ie, no disagreements).</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>These analyses indicate that the primary findings were robust to alternative GT tie handling and to reasonable restrictions based on reviewer agreement. More favorable results in the full-consensus subset are consistent with these cases representing clearer, less ambiguous presentations, whereas the primary plurality-based GT approach intentionally retained borderline and clinically variable cases in the evaluation.</p>
      </sec>
      <sec>
        <title>Secondary Endpoints</title>
        <p>Secondary Endpoints evaluated the clinical relevance of PHA outputs beyond level-of-care assignment, including possible conditions and recommendations related to escalation, self-care, and labs and testing. These analyses were conducted in the modified analysis population because such outputs were only generated when PHA recommended nonurgent follow-up care or self-care.</p>
        <p>For possible conditions, text-based overlap between PHA outputs and grader-provided condition lists showed moderate conceptual alignment. Across the modified analysis population, the mean GT list length was 3.1 conditions per case, and most cases (approximately 70%) had 2 to 4 GT conditions. Across reviewer-case pairs, the hit rate was 0.76, recall was 0.42, precision was 0.26, and <italic>F</italic><sub>1</sub>-score was 0.31. A high hit rate (76%) indicates that in most cases PHA includes at least one condition aligned with each grader’s thinking. Recall (42%) shows that PHA captures a substantial—but not exhaustive—portion of grader-specified conditions. Precision (26%) is inherently constrained by the fact that PHA always provides five conditions, while GT lists, on average, just over three. PHA commonly proposes additional companion or adjacent conditions beyond what any single reviewer lists. <italic>F</italic><sub>1</sub>-score (0.31) reflected this balance of partial recall and modest precision.</p>
        <p>Among the 377 cases in the modified analysis population, 376 (99.7%) received a positive majority vote indicating that graders agreed the PHA list included at least 1 clinically appropriate condition (95% bootstrap CI 0.992-1.000). This measure reflects physician review of the presented PHA conditions list rather than an independently generated list, and therefore should be interpreted as supportive evidence of clinical appropriateness rather than as an independent accuracy metric.</p>
        <p>A similar agreement-based approach was used to evaluate the appropriateness of PHA recommendation outputs beyond possible conditions. Agreement with each recommendation type was rated by physician reviewers on a 5-point Likert scale (strongly disagree, disagree, neutral, agree, and strongly agree). Ratings were grouped as disagree (1-2), neutral (3), and agree (4-5). <xref ref-type="table" rid="table6">Table 6</xref> shows the mean Likert score with 95% CI and the number and percentage of ratings in each grouped category. Agreement was highest for escalation and self-care guidance and lower, but still favorable overall, for lab and testing recommendations. Mean Likert scores were 4.41 (95% CI 4.37-4.44) for escalation recommendations, 4.37 (95% CI 4.33-4.41) for self-care recommendations, and 3.99 (95% CI 3.93-4.03) for lab and testing recommendations; corresponding proportions of favorable ratings (Likert 4-5) were 93.9%, 91.4%, and 77.0%, respectively.</p>
        <table-wrap position="float" id="table6">
          <label>Table 6</label>
          <caption>
            <p>Agreement with Personal Health Assistant escalation recommendations, self-care recommendations, and lab and test recommendations.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="210"/>
            <col width="110"/>
            <col width="200"/>
            <col width="190"/>
            <col width="140"/>
            <col width="150"/>
            <thead>
              <tr valign="top">
                <td>Recommendation type</td>
                <td>Ratings, n</td>
                <td>Likert score, mean (95% CI)</td>
                <td>Disagree (1-2), n (%)</td>
                <td>Neutral (3), n (%)</td>
                <td>Agree (4-5), n (%)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Escalation recommendation</td>
                <td>1860</td>
                <td>4.41 (4.37-4.44)</td>
                <td>61 (3.3)</td>
                <td>53 (2.8)</td>
                <td>1746 (93.9)</td>
              </tr>
              <tr valign="top">
                <td>Self-care recommendation</td>
                <td>1859</td>
                <td>4.37 (4.33-4.41)</td>
                <td>79 (4.2)</td>
                <td>81 (4.4)</td>
                <td>1699 (91.4)</td>
              </tr>
              <tr valign="top">
                <td>Lab and testing recommendation</td>
                <td>1843</td>
                <td>3.99 (3.93-4.03)</td>
                <td>263 (14.3)</td>
                <td>161 (8.7)</td>
                <td>1419 (77.0)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal PHA Findings</title>
        <p>This study used a structured predeployment analytical validation approach to evaluate PHA, a patient-facing AI system for symptom guidance. The principal finding was that PHA met the prespecified overtriage criterion but did not meet the prespecified criteria for urgent and nonurgent undertriage. The main implication is that unnecessary escalation was relatively limited under the study conditions, but the system still assigned lower-acuity recommendations too often among cases judged to require urgent or nonurgent clinician follow-up. The key performance gap was therefore undertriage rather than overtriage. This is an early evidentiary step for identifying performance gaps under controlled conditions before prospective clinical evaluation. In this case, the results provide product-specific evidence to guide further refinement of the evaluated PHA configuration.</p>
        <p>Additionally, when generated, PHA’s recommendation content (escalation recommendations, self-care recommendations, and lab and testing recommendations) was generally viewed by physician reviewers as clinically appropriate, and the possible conditions output showed partial alignment with physician-generated condition lists even though it did not reproduce those lists exactly. These secondary findings should be interpreted as supportive evidence regarding response quality beyond the primary level-of-care recommendations.</p>
      </sec>
      <sec>
        <title>Principal Methodological Contributions</title>
        <p>These findings should be understood in the context of staged evaluation for patient-facing clinical guidance tools. Fraser and colleagues [<xref ref-type="bibr" rid="ref20">20</xref>] argue that such tools should first undergo predeployment laboratory testing and then progress to observational and randomized studies in real-world settings. SCARF further emphasizes transparent reporting of case selection, reference-standard assignment, evaluation design, and outcomes [<xref ref-type="bibr" rid="ref12">12</xref>]. Consistent with these principles, the present study represents one structured early-stage evaluation of a locked PHA configuration using prespecified clinically relevant error Endpoints, simulated interactive encounters, expert physician review, and explicit ground-truth derivation rules.</p>
        <p>Several design choices are central to this contribution. In remote, exam-free settings, the most consequential immediate output is whether the patient is advised to seek care. Accordingly, this study prioritized prespecified urgent undertriage, nonurgent undertriage, and overtriage Endpoints rather than diagnostic accuracy as the primary measure of performance. Our evaluation used clinically structured synthetic cases rooted in established nurse triage protocols, with deliberate coverage of urgent, nonurgent, and self-care scenarios. A key strength was its broad clinical coverage, with cases spanning multiple presenting-symptom categories and diverse patient personas, including scenarios in which acuity depended on medical history, scenarios in which urgency was clarified only through elicited conversational detail, and scenarios in which the presenting symptom was largely unrelated to the available historical context. Finally, GT was based on aggregated blinded judgments rather than a single reviewer, allowing the evaluation to better reflect the inter-reviewer variability that naturally arises in clinically ambiguous and borderline cases. Supportive analyses reinforced this design logic: the overall conclusion did not materially change under alternative GT tie handling, indicating that the primary findings were not driven by a single aggregation assumption. Reviewer-specific mistriage rates also varied, with urgent undertriage ranging from 0% to 16%, nonurgent undertriage from 0% to 41%, and overtriage from 4% to 17% across individual reviewers. The observed spread reinforces that triage assessment is inherently noisy and supports the use of aggregated responses from a panel of physicians as a more robust comparator than 1-2 reviewers.</p>
        <p>An additional methodological feature of this study was the use of a multiturn encounter framework rather than static summaries alone, consistent with prior work [<xref ref-type="bibr" rid="ref11">11</xref>] emphasizing that patient-facing symptom-guidance tools should be evaluated as interactive systems in which clinically relevant information emerges through active elicitation. Symptom severity and other urgency-relevant features may become apparent only through iterative questioning rather than being fully specified at the outset. For a patient-facing symptom-guidance product, the purpose of that interaction is not only to surface plausible conditions, but also to elicit the urgency-relevant information needed to support an appropriate level-of-care recommendation.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>While this study provides an early evaluation of an AI-based symptom guidance system, interpretation of primary and secondary Endpoints should account for the limitations inherent in synthetic data.</p>
        <p>First, the use of synthetic cases may introduce spectrum and representational bias because such datasets can overrepresent more prototypical presentations and may not fully capture atypical, incomplete, noisy, or culturally and linguistically diverse symptom descriptions encountered in real-world use. Although case generation used structured persona variation and independent physician quality-control review, no formal quantitative demographic or intersectional bias audit was conducted. Notable, however, the dataset was not limited to uniformly clear cases. The more favorable performance observed in the full-consensus subset suggests that cases on which all reviewers agreed represented a clearer subset, whereas the broader dataset included more borderline presentations that generated reviewer disagreement. Second, the use of a single published telephone-triage source also shaped the clinical pathways and urgency assumptions represented in the case set. This source was relevant to the product’s intended use in remote triage and a general nonacute setting, although it may not reflect all clinical practices within such settings.</p>
        <p>Similarly, although the patient simulator allows for more naturalistic exchanges than static or prescripted vignettes, it cannot fully capture the variability of real-world encounters. Patient communication is influenced by health literacy and cultural, cognitive, psychosocial, and situational factors, and users may omit information, use unexpected language, or change their description during an interaction. In addition, the simulator was grounded in predefined synthetic source cases; therefore, any errors or structural assumptions in the case data could propagate through the simulated interaction.</p>
        <p>A further limitation is that the evaluated PHA configuration was supplied with rich historical medical record information that may exceed what would be available in some real-world use settings (eg, prior patient encounter record recorded in the last 12 months). This creates the possibility of optimistic performance bias due to enriched input information, such that estimated triage performance may be better than would be observed when less contextual information is available. In a real primary care population, for instance, a young patient may not have seen a physician in the last 12 months; however, in this study all user-simulations included this past visit information.</p>
        <p>Finally, AV does not establish accuracy against clinical outcomes (eg, patient disposition), real-world safety, effectiveness, or clinical utility. The 400:300:100 urgency allocation was deliberately Endpoint–enriched to provide adequate primary Endpoint denominators and was not intended to represent prevalence in any deployment setting. Urgent and nonurgent undertriage were estimated separately within cases assigned urgent and nonurgent GT, respectively; therefore, these Endpoint–specific conditional rates are not directly determined by the relative prevalence of the 3 urgency categories in the study sample. However, they may still vary if the clinical case mix within an urgency category differs across settings. In addition, because urgency prevalence may differ across direct-to-consumer, primary care, urgent care, emergency, and prehospital populations, aggregate error rates, predictive values, escalation frequency, and operational burden may also differ in practice. Future prospective studies should therefore use a case distribution representative of the intended deployment setting or apply appropriate prevalence-based standardization or reweighting. Such studies should assess actual user interactions, clinical outcomes, utilization, and comparisons with existing triage workflows.</p>
      </sec>
      <sec>
        <title>Conclusion</title>
        <p>This study provides a controlled predeployment assessment of a locked patient-facing symptom-guidance AI configuration. PHA met the prespecified overtriage criterion but did not meet the urgent or nonurgent undertriage criteria. The study also illustrates one approach to early-stage evaluation using simulated multiturn encounters, prespecified directional error Endpoints, and blinded physician-panel adjudication. Further product refinement and prospective evaluation in the intended-use population are needed before broader clinical use.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Examples of information provided to the patient simulator and information provided directly to Personal Health Assistant (PHA).</p>
        <media xlink:href="formative_v10i1e103419_app1.docx" xlink:title="DOCX File , 3814 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Description of case generation process.</p>
        <media xlink:href="formative_v10i1e103419_app2.docx" xlink:title="DOCX File , 3820 KB"/>
      </supplementary-material>
      <supplementary-material id="app3">
        <label>Multimedia Appendix 3</label>
        <p>Enriched symptoms. The collection of test cases were enriched for the following urgent and nonurgent symptoms identified as being important to triage correctly.</p>
        <media xlink:href="formative_v10i1e103419_app3.docx" xlink:title="DOCX File , 3817 KB"/>
      </supplementary-material>
      <supplementary-material id="app4">
        <label>Multimedia Appendix 4</label>
        <p>Review instructions and reviewer workflow.</p>
        <media xlink:href="formative_v10i1e103419_app4.docx" xlink:title="DOCX File , 5196 KB"/>
      </supplementary-material>
      <supplementary-material id="app5">
        <label>Multimedia Appendix 5</label>
        <p>Statistical analysis details.</p>
        <media xlink:href="formative_v10i1e103419_app5.docx" xlink:title="DOCX File , 3815 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">ACS-COT</term>
          <def>
            <p>American College of Surgeons Committee on Trauma</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">AV</term>
          <def>
            <p>analytical validation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">EHR</term>
          <def>
            <p>electronic health record</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">GT</term>
          <def>
            <p>ground truth</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">HIE</term>
          <def>
            <p>health information exchange</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">PHA</term>
          <def>
            <p>Personal Health Assistant</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">SCARF</term>
          <def>
            <p>Symptom Checker Accuracy Reporting Framework</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors would like to thank Andrew Trister, Adrian Sanchez, Lawrence Deang, Sam Pugh, Sarah Pozniak, Julia Saiz, and Claudia Amar for their support in execution of this study and preparation of this manuscript. OpenAI’s ChatGPT, using the GPT-4o model, was used to assist with creation of the manuscript figures. OpenAI’s ChatGPT, using the GPT-5.6 Thinking model, was used to review the manuscript for internal consistency with the prespecified statistical analysis plan and the reported statistical analysis results. The authors critically reviewed the AI-assisted output, independently verified the manuscript content and statistical results, and take full responsibility for the final manuscript.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>The datasets generated and analyzed during this study are available from the corresponding author on reasonable request.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>This study was funded by Verily Health. All authors were employees of Verily Health during their contributions to the study. Verily Health was involved in the study design, execution, analysis, interpretation of the results, and preparation of the manuscript.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>AP led the study design, synthetic case generation, overall study execution, analysis of study results, and manuscript drafting. PB provided statistical review of results. NV, BM, and NL deployed the PHA product and set up the infrastructure to simulate patient conversations and to assess PHA with the synthetic cases. VP provided operational support for execution of the study. PV provided clinical review and manuscript contributions. LL contributed to study design, analysis of results, and manuscript drafting.</p>
      </fn>
      <fn fn-type="conflict">
        <p>All authors were employees of Verily Health when this work was conducted and held equity in the company as part of their employment. Verily Health developed the investigational Personal Health Assistant evaluated in this study and funded the study.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mentzou</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Rogers</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Carvalho</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Daly</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Malone</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kerasidou</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence in digital self-diagnosis tools: a narrative overview of reviews</article-title>
          <source>Mayo Clin Proc Digit Health</source>
          <year>2025</year>
          <month>09</month>
          <volume>3</volume>
          <issue>3</issue>
          <fpage>100242</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2949-7612(25)00049-5"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.mcpdig.2025.100242</pub-id>
          <pub-id pub-id-type="medline">40686623</pub-id>
          <pub-id pub-id-type="pii">S2949-7612(25)00049-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC12271431</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kopka</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>von Kalckreuth</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Feufel</surname>
              <given-names>MA</given-names>
            </name>
          </person-group>
          <article-title>Accuracy of online symptom assessment applications, large language models, and laypeople for self-triage decisions</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>03</month>
          <day>25</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>178</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01566-6</pub-id>
          <pub-id pub-id-type="medline">40133390</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01566-6</pub-id>
          <pub-id pub-id-type="pmcid">PMC11937345</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hammoud</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Douglas</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Darmach</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Alawneh</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sanyal</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kanbour</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the diagnostic performance of symptom checkers: clinical vignette study</article-title>
          <source>JMIR AI</source>
          <year>2024</year>
          <month>04</month>
          <day>29</day>
          <volume>3</volume>
          <fpage>e46875</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://ai.jmir.org/2024//e46875/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/46875</pub-id>
          <pub-id pub-id-type="medline">38875676</pub-id>
          <pub-id pub-id-type="pii">v3i1e46875</pub-id>
          <pub-id pub-id-type="pmcid">PMC11091811</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gilbert</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mehl</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Baluch</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cawley</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Challiner</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Fraser</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Millen</surname>
              <given-names>Elizabeth</given-names>
            </name>
            <name name-style="western">
              <surname>Montazeri</surname>
              <given-names>Maryam</given-names>
            </name>
            <name name-style="western">
              <surname>Multmeier</surname>
              <given-names>Jan</given-names>
            </name>
            <name name-style="western">
              <surname>Pick</surname>
              <given-names>Fiona</given-names>
            </name>
            <name name-style="western">
              <surname>Richter</surname>
              <given-names>Claudia</given-names>
            </name>
            <name name-style="western">
              <surname>Türk</surname>
              <given-names>Ewelina</given-names>
            </name>
            <name name-style="western">
              <surname>Upadhyay</surname>
              <given-names>Shubhanan</given-names>
            </name>
            <name name-style="western">
              <surname>Virani</surname>
              <given-names>Vishaal</given-names>
            </name>
            <name name-style="western">
              <surname>Vona</surname>
              <given-names>Nicola</given-names>
            </name>
            <name name-style="western">
              <surname>Wicks</surname>
              <given-names>Paul</given-names>
            </name>
            <name name-style="western">
              <surname>Novorol</surname>
              <given-names>Claire</given-names>
            </name>
          </person-group>
          <article-title>How accurate are digital symptom assessment apps for suggesting conditions and urgency advice? A clinical vignettes comparison to GPs</article-title>
          <source>BMJ Open</source>
          <year>2020</year>
          <month>12</month>
          <day>16</day>
          <volume>10</volume>
          <issue>12</issue>
          <fpage>e040269</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmjopen.bmj.com/lookup/pmidlookup?view=long&#38;pmid=33328258"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmjopen-2020-040269</pub-id>
          <pub-id pub-id-type="medline">33328258</pub-id>
          <pub-id pub-id-type="pii">bmjopen-2020-040269</pub-id>
          <pub-id pub-id-type="pmcid">PMC7745523</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Semigran</surname>
              <given-names>HL</given-names>
            </name>
            <name name-style="western">
              <surname>Linder</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Gidengil</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mehrotra</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of symptom checkers for self diagnosis and triage: audit study</article-title>
          <source>BMJ</source>
          <year>2015</year>
          <month>07</month>
          <day>08</day>
          <volume>351</volume>
          <fpage>h3480</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.bmj.com/lookup/pmidlookup?view=long&#38;pmid=26157077"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmj.h3480</pub-id>
          <pub-id pub-id-type="medline">26157077</pub-id>
          <pub-id pub-id-type="pmcid">PMC4496786</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Schmieding</surname>
              <given-names>ML</given-names>
            </name>
            <name name-style="western">
              <surname>Kopka</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Schmidt</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Schulz-Niethammer</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Balzer</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Feufel</surname>
              <given-names>MA</given-names>
            </name>
          </person-group>
          <article-title>Triage accuracy of symptom checker apps: 5-year follow-up evaluation</article-title>
          <source>J Med Internet Res</source>
          <year>2022</year>
          <month>05</month>
          <day>10</day>
          <volume>24</volume>
          <issue>5</issue>
          <fpage>e31810</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2022/5/e31810/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/31810</pub-id>
          <pub-id pub-id-type="medline">35536633</pub-id>
          <pub-id pub-id-type="pii">v24i5e31810</pub-id>
          <pub-id pub-id-type="pmcid">PMC9131144</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wallace</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Chan</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chidambaram</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hanna</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Iqbal</surname>
              <given-names>FM</given-names>
            </name>
            <name name-style="western">
              <surname>Acharya</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Normahani</surname>
              <given-names>Pasha</given-names>
            </name>
            <name name-style="western">
              <surname>Ashrafian</surname>
              <given-names>Hutan</given-names>
            </name>
            <name name-style="western">
              <surname>Markar</surname>
              <given-names>Sheraz R</given-names>
            </name>
            <name name-style="western">
              <surname>Sounderajah</surname>
              <given-names>Viknesh</given-names>
            </name>
            <name name-style="western">
              <surname>Darzi</surname>
              <given-names>Ara</given-names>
            </name>
          </person-group>
          <article-title>The diagnostic and triage accuracy of digital and online symptom checker tools: a systematic review</article-title>
          <source>NPJ Digit Med</source>
          <year>2022</year>
          <month>08</month>
          <day>17</day>
          <volume>5</volume>
          <issue>1</issue>
          <fpage>118</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-022-00667-w</pub-id>
          <pub-id pub-id-type="medline">35977992</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-022-00667-w</pub-id>
          <pub-id pub-id-type="pmcid">PMC9385087</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jia</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Qiao</surname>
              <given-names>Youlin</given-names>
            </name>
          </person-group>
          <article-title>Independent and collaborative performance of large language models and healthcare professionals in diagnosis and triage</article-title>
          <source>NPJ Digit Med</source>
          <year>2026</year>
          <month>02</month>
          <day>06</day>
          <volume>9</volume>
          <issue>1</issue>
          <fpage>222</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-026-02409-8</pub-id>
          <pub-id pub-id-type="medline">41652180</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-026-02409-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC12992554</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ramaswamy</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Tyagi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Hugo</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jayaraman</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Jangda</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Te</surname>
              <given-names>Alexis E</given-names>
            </name>
            <name name-style="western">
              <surname>Kaplan</surname>
              <given-names>Steven A</given-names>
            </name>
            <name name-style="western">
              <surname>Lampert</surname>
              <given-names>Joshua</given-names>
            </name>
            <name name-style="western">
              <surname>Freeman</surname>
              <given-names>Robert</given-names>
            </name>
            <name name-style="western">
              <surname>Gavin</surname>
              <given-names>Nicholas</given-names>
            </name>
            <name name-style="western">
              <surname>Tewari</surname>
              <given-names>Ashutosh K</given-names>
            </name>
            <name name-style="western">
              <surname>Sakhuja</surname>
              <given-names>Ankit</given-names>
            </name>
            <name name-style="western">
              <surname>Naved</surname>
              <given-names>Bilal</given-names>
            </name>
            <name name-style="western">
              <surname>Charney</surname>
              <given-names>Alexander W</given-names>
            </name>
            <name name-style="western">
              <surname>Omar</surname>
              <given-names>Mahmud</given-names>
            </name>
            <name name-style="western">
              <surname>Gorin</surname>
              <given-names>Michael A</given-names>
            </name>
            <name name-style="western">
              <surname>Klang</surname>
              <given-names>Eyal</given-names>
            </name>
            <name name-style="western">
              <surname>Nadkarni</surname>
              <given-names>Girish N</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT Health performance in a structured test of triage recommendations</article-title>
          <source>Nat Med</source>
          <year>2026</year>
          <month>05</month>
          <volume>32</volume>
          <issue>5</issue>
          <fpage>1671</fpage>
          <lpage>1675</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-026-04297-7</pub-id>
          <pub-id pub-id-type="medline">41731097</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-026-04297-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC13190235</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kopka</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Feufel</surname>
              <given-names>MA</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the accuracy of chatGPT model versions for giving care-seeking advice</article-title>
          <source>Commun Med (Lond)</source>
          <year>2026</year>
          <volume>6</volume>
          <issue>1</issue>
          <fpage>171</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s43856-026-01466-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s43856-026-01466-0</pub-id>
          <pub-id pub-id-type="medline">41735505</pub-id>
          <pub-id pub-id-type="pii">10.1038/s43856-026-01466-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC13031804</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Breda</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yousif</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hawkins</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Cotoi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>R</given-names>
            </name>
            <collab>et al</collab>
          </person-group>
          <article-title>SymptomAI: toward a conversational AI agent for everyday symptom assessment</article-title>
          <source>arXiv. Preprint posted online on May 5, 2026</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2605.04012</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kopka</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Feufel</surname>
              <given-names>MA</given-names>
            </name>
          </person-group>
          <article-title>How to evaluate the accuracy of symptom checkers and diagnostic decision support systems: Symptom Checker Accuracy Reporting Framework (SCARF)</article-title>
          <source>JMIR Hum Factors</source>
          <year>2026</year>
          <month>01</month>
          <day>16</day>
          <volume>13</volume>
          <fpage>e76168</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://humanfactors.jmir.org/2026//e76168/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/76168</pub-id>
          <pub-id pub-id-type="medline">41544248</pub-id>
          <pub-id pub-id-type="pii">v13i1e76168</pub-id>
          <pub-id pub-id-type="pmcid">PMC12810947</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Briggs</surname>
              <given-names>JK</given-names>
            </name>
          </person-group>
          <source>Telephone Triage Protocols for Nurses</source>
          <year>2021</year>
          <publisher-loc>Philadelphia, PA</publisher-loc>
          <publisher-name>Wolters Kluwer</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rashidian</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Amar</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Pugh</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>E</given-names>
            </name>
            <collab>et al</collab>
          </person-group>
          <article-title>AI agents for conversational patient triage: preliminary simulation-based evaluation with real-world EHR data</article-title>
          <source>arXiv:2506.04032. Preprint posted online on June 4, 2025</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2506.04032</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="web">
          <article-title>HumanSignal / label-studio</article-title>
          <source>GitHub</source>
          <year>2026</year>
          <access-date>2026-07-30</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/HumanSignal/label-studio">https://github.com/HumanSignal/label-studio</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="web">
          <article-title>Safety and effectiveness of remote pre-hospital triage for appropriate emergency department attendances and service use: an evidence review</article-title>
          <source>Health Research Board</source>
          <access-date>2026-07-31</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.hrb.ie/publication/safety-and-effectiveness-of-remote-pre-hospital-triage-for-appropriate-emergency-department-attendances-and-service-use/">https://www.hrb.ie/publication/safety-and-effectiveness-of-remote-pre-hospital-triage-for-appropriate-emergency-department-attendances-and-service-use/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Graversen</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Christensen</surname>
              <given-names>MB</given-names>
            </name>
            <name name-style="western">
              <surname>Pedersen</surname>
              <given-names>AF</given-names>
            </name>
            <name name-style="western">
              <surname>Carlsen</surname>
              <given-names>AH</given-names>
            </name>
            <name name-style="western">
              <surname>Bro</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Christensen</surname>
              <given-names>HC</given-names>
            </name>
            <name name-style="western">
              <surname>Vestergaard</surname>
              <given-names>C H</given-names>
            </name>
            <name name-style="western">
              <surname>Huibers</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Safety, efficiency and health-related quality of telephone triage conducted by general practitioners, nurses, or physicians in out-of-hours primary care: a quasi-experimental study using the Assessment of Quality in Telephone Triage (AQTT) to assess audio-recorded telephone calls</article-title>
          <source>BMC Fam Pract</source>
          <year>2020</year>
          <month>05</month>
          <day>09</day>
          <volume>21</volume>
          <issue>1</issue>
          <fpage>84</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcfampract.biomedcentral.com/articles/10.1186/s12875-020-01122-z"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12875-020-01122-z</pub-id>
          <pub-id pub-id-type="medline">32386511</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12875-020-01122-z</pub-id>
          <pub-id pub-id-type="pmcid">PMC7211335</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cotte</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Mueller</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Gilbert</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Blümke</surname>
              <given-names>Bibiana</given-names>
            </name>
            <name name-style="western">
              <surname>Multmeier</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Hirsch</surname>
              <given-names>MC</given-names>
            </name>
            <name name-style="western">
              <surname>Wicks</surname>
              <given-names>Paul</given-names>
            </name>
            <name name-style="western">
              <surname>Wolanski</surname>
              <given-names>Joseph</given-names>
            </name>
            <name name-style="western">
              <surname>Tutschkow</surname>
              <given-names>Darja</given-names>
            </name>
            <name name-style="western">
              <surname>Schade Brittinger</surname>
              <given-names>Carmen</given-names>
            </name>
            <name name-style="western">
              <surname>Timmermann</surname>
              <given-names>Lars</given-names>
            </name>
            <name name-style="western">
              <surname>Jerrentrup</surname>
              <given-names>Andreas</given-names>
            </name>
          </person-group>
          <article-title>Safety of triage self-assessment using a symptom assessment app for walk-in patients in the emergency care setting: observational prospective cross-sectional study</article-title>
          <source>JMIR Mhealth Uhealth</source>
          <year>2022</year>
          <month>03</month>
          <day>28</day>
          <volume>10</volume>
          <issue>3</issue>
          <fpage>e32340</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mhealth.jmir.org/2022/3/e32340/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/32340</pub-id>
          <pub-id pub-id-type="medline">35343909</pub-id>
          <pub-id pub-id-type="pii">v10i3e32340</pub-id>
          <pub-id pub-id-type="pmcid">PMC9002590</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gellert</surname>
              <given-names>GA</given-names>
            </name>
            <name name-style="western">
              <surname>Kuszczyński</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Marcjasz</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Jaszczak</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Price</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Orzechowski</surname>
              <given-names>PM</given-names>
            </name>
          </person-group>
          <article-title>A comparative performance analysis of live clinical triage using rules-based triage protocols versus artificial intelligence-based automated virtual triage</article-title>
          <source>J Hosp Adm</source>
          <year>2023</year>
          <volume>13</volume>
          <issue>1</issue>
          <fpage>8</fpage>
          <pub-id pub-id-type="doi">10.5430/jha.v13n1p8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Fraser</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Coiera</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Safety of patient-facing digital symptom checkers</article-title>
          <source>Lancet</source>
          <year>2018</year>
          <month>11</month>
          <day>24</day>
          <volume>392</volume>
          <issue>10161</issue>
          <fpage>2263</fpage>
          <lpage>2264</lpage>
          <pub-id pub-id-type="doi">10.1016/S0140-6736(18)32819-8</pub-id>
          <pub-id pub-id-type="medline">30413281</pub-id>
          <pub-id pub-id-type="pii">S0140-6736(18)32819-8</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
