<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e100423</article-id><article-id pub-id-type="doi">10.2196/100423</article-id><article-categories><subj-group subj-group-type="heading"><subject>Viewpoint</subject></subj-group></article-categories><title-group><article-title>A Clinician-Centered Evaluation Framework for Large Language Models in Patient Education: Integrating the Technology Acceptance Model and Medical Condition Regard Scale</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Austria</surname><given-names>Davis</given-names></name><degrees>DNP, MBA, MSN</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Williams</surname><given-names>Grace Lord</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Girardo</surname><given-names>Christopher</given-names></name><degrees>DO, MHI</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Arowolo</surname><given-names>Micheal Olaolu</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mishra</surname><given-names>Meenakshi</given-names></name><degrees>PhD, MPH, MSc</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pope</surname><given-names>Charlene</given-names></name><degrees>PhD, MPH</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Axon</surname><given-names>R Neal</given-names></name><degrees>MD, MSCR</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hill</surname><given-names>Jason Bradley</given-names></name><degrees>MD, MS, MMM</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff6">6</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Public Health Sciences, College of Arts and Sciences, Xavier University of Louisiana</institution><addr-line>1 Drexel Drive</addr-line><addr-line>New Orleans</addr-line><addr-line>LA</addr-line><country>United States</country></aff><aff id="aff2"><institution>Ochsner Center for Outcomes and Health Services Research, Ochsner Health System</institution><addr-line>New Orleans</addr-line><addr-line>LA</addr-line><country>United States</country></aff><aff id="aff3"><institution>Xavier Ochsner College of Medicine</institution><addr-line>New Orleans</addr-line><addr-line>LA</addr-line><country>United States</country></aff><aff id="aff4"><institution>Department of Pathology, Ochsner Health System</institution><addr-line>New Orleans</addr-line><addr-line>LA</addr-line><country>United States</country></aff><aff id="aff5"><institution>HEROIC, Ralph H Johnson VA Health Care System</institution><addr-line>Charleston</addr-line><addr-line>SC</addr-line><country>United States</country></aff><aff id="aff6"><institution>Hospital Medicine and Innovation, Ochsner Health System</institution><addr-line>New Orleans</addr-line><addr-line>LA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>MacNeill</surname><given-names>Luke</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Shao</surname><given-names>Lex</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Davis Austria, DNP, MBA, MSN, Department of Public Health Sciences, College of Arts and Sciences, Xavier University of Louisiana, 1 Drexel Drive, New Orleans, 70125, LA, United States, 1 917-774-7740; <email>daustria@xula.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>9</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e100423</elocation-id><history><date date-type="received"><day>06</day><month>05</month><year>2026</year></date><date date-type="rev-recd"><day>22</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>25</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Davis Austria, Grace Lord Williams, Christopher Girardo, Micheal Olaolu Arowolo, Meenakshi Mishra, Charlene Pope, R Neal Axon, Jason Bradley Hill. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 21.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e100423"/><abstract><p>Inadequate postcare patient education contributes to preventable readmissions and adverse outcomes that disproportionately affect medically complex, high-need communities. Large language models (LLMs) show promise for generating personalized, plain-language patient education at scale. However, existing LLM evaluation frameworks prioritize technical accuracy over patient accessibility and alignment with health literacy, and few explicitly account for the attitudinal influences that clinician evaluators may introduce into the rating process. In this viewpoint, we introduce an evaluation framework that pairs the Technology Acceptance Model (TAM) with the Medical Condition Regard Scale (MCRS). We call it the TAM-MCRS LLM evaluation framework, a novel clinician-centered approach for comparing which LLMs produce the highest-quality postcare patient education across accuracy, appropriateness, clarity, and completeness. We intend for this framework to be used to evaluate LLM-generated patient education outputs through a 2-arm design that pairs an expert clinician panel with automated assessment methods, allowing for interarm comparison using clinical vignettes while accounting for measured evaluator attitudinal variance. The framework was developed through the National Institutes of Health Artificial Intelligence/Machine Learning Consortium to Advance Health Equity and Researcher Diversity (AIM-AHEAD) Clinicians Leading Ingenuity IN AI Quality (CLINAQ) fellowship program, in partnership with Ochsner Health and Xavier University of Louisiana. The TAM-MCRS framework integrates two theoretical lenses. TAM maps perceived usefulness onto accuracy and completeness, and perceived ease of use onto clarity and appropriateness. Clinicians rate each output with a TAM-based questionnaire, and we then administer the MCRS as a postscoring attitudinal covariate to see whether their regard for the conditions represented in the vignettes influences those ratings. Together, the 2 lenses are intended to produce evidence that is objective, theoretically grounded, clinically realistic, and disparity-responsive. Implications for clinician informaticists, health system governance, and responsible AI deployment are discussed. This viewpoint reflects the authors&#x2019; position and is written for clinician informaticists, health system AI governance leaders, implementation scientists, and investigators evaluating LLM-generated patient education.</p></abstract><kwd-group><kwd>large language models</kwd><kwd>patient education</kwd><kwd>technology acceptance model</kwd><kwd>Medical Condition Regard Scale</kwd><kwd>clinical informatics</kwd><kwd>health informatics</kwd><kwd>health equity</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>health literacy</kwd><kwd>readability</kwd></kwd-group></article-meta></front><body><sec id="s1"><title>Large Language Models in Patient Education: Opportunity and Persistent Gaps</title><p>Clinicians, broadly defined in this paper as physicians, advanced practice providers, and nurses involved in patient education and discharge planning, serve as primary translators between medical information and patient understanding. As AI tools increasingly enter clinical environments, health care professionals and clinical informaticists face a pressing question: how do we evaluate whether these tools actually serve our patients, not just in terms of technical accuracy, but also in accessibility, relevance, and responsiveness to diverse patient needs?</p><p>The use of large language models (LLMs) in patient-facing applications has grown substantially since the release of publicly accessible generative AI tools. In clinical settings, early use cases have included generating postcare patient education, answering patient portal questions, and producing condition-specific discharge materials. These applications are particularly relevant to clinical practice, where patient education is both a professional standard and a time-intensive responsibility. Inadequate postcare patient education contributes to medication errors, preventable readmissions, and adverse outcomes, with the average cost of a single avoidable readmission estimated at approximately US $15,200 and potentially avoidable emergency department encounters representing approximately US $64.4 billion annually in the United States [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>Research has demonstrated that LLMs can produce patient education content at a level comparable to, and in some cases exceeding, the quality of standard institutional materials [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. However, a critical limitation emerges consistently: most LLM-generated health content defaults to literacy levels that exceed national reading averages [<xref ref-type="bibr" rid="ref6">6</xref>]. Studies have found that AI-generated patient materials frequently score at a 10th-grade reading level or higher, while health literacy guidelines recommend content at a 6th-grade level for broad accessibility [<xref ref-type="bibr" rid="ref7">7</xref>]. For patients with limited formal education, older adults, or those navigating health crises, this gap translates directly into reduced comprehension and poorer health outcomes [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Equity concerns extend beyond readability. The training data underlying most general-purpose LLMs overrepresent White, English-speaking, and formally educated populations [<xref ref-type="bibr" rid="ref9">9</xref>]. LLM outputs may therefore reflect cultural assumptions, communication styles, and health beliefs that do not align with the lived experiences of patients from racial and ethnic minority groups, those from lower socioeconomic backgrounds, those with less education, or those navigating language barriers [<xref ref-type="bibr" rid="ref10">10</xref>]. These barriers reflect what has been described as algorithmic bias by omission, a concept central to the growing literature on health AI [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. These gaps are particularly consequential in regions such as the Louisiana Diabetes Belt and Cancer Alley, where Black, Hispanic, and older adult patients carry disproportionate rates of chronic illness and face compounding barriers to clear health communication [<xref ref-type="bibr" rid="ref13">13</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. In algorithmic development, including these patient demographics alongside conditions and comorbidities is necessary if training corpora are to represent the population they serve.</p><p>Our aim in this Viewpoint is to demonstrate how clinician evaluation of LLM-generated patient education should account not only for output quality but also for clinicians&#x2019; attitudinal regard toward the medical conditions represented in evaluation scenarios and to propose, as one conceptual approach, a framework pairing the Technology Acceptance Model (TAM) with the Medical Condition Regard Scale (MCRS), which we call the TAM-MCRS LLM evaluation framework. The framework was developed through the National Institutes of Health (NIH) Artificial Intelligence/Machine Learning Consortium to Advance Health Equity and Researcher Diversity (AIM-AHEAD) Clinicians Leading Ingenuity IN AI Quality (CLINAQ) fellowship program at Xavier University of Louisiana in partnership with Ochsner Health, with foundational conceptualization supported through the Veterans Affairs Quality Scholars fellowship program. Within the fellowship, the framework was designed to evaluate the performance of LLMs in generating high-quality postcare patient education across accuracy, appropriateness, clarity, and completeness, using clinical vignettes and evaluation by an expert clinician panel. We write for clinician informaticists, health system AI governance leaders, implementation scientists, and investigators evaluating LLM-generated patient education.</p><p>The remainder of this Viewpoint proceeds in 3 moves. We first set out the 2 lenses the framework rests on, explaining what TAM contributes to judging whether patient education is usable and what the MCRS contributes to judging the clinicians&#x2019; attitudinal variation. We then describe why the two belong together and what neither achieves alone. Finally, we present the framework itself, a 2-arm design in which clinician and automated evaluation are applied to the same outputs and compared, and we close with what this means for practice, health system governance, and the discipline.</p></sec><sec id="s2"><title>The Foundation of Our Framework: Benefits and Limitations</title><sec id="s2-1"><title>The TAM: A Theoretically Grounded Lens for Technology Adoption</title><p>TAM, originally proposed by Davis in 1989 [<xref ref-type="bibr" rid="ref18">18</xref>] and subsequently extended by Venkatesh et al [<xref ref-type="bibr" rid="ref19">19</xref>], has become one of the most frequently applied frameworks in health informatics. Historically used to evaluate the adoption of electronic health records, patient portals, telehealth platforms, and clinical decision support systems, this scope of use has since expanded to evaluations of AI tools in clinical settings [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. While TAM&#x2019;s constructs of perceived usefulness (PU) and perceived ease of use (PEOU) provide a solid foundation for understanding technology adoption, TAM posits that PU and PEOU determine only the behavioral intention to adopt a technology. Therefore, their application to LLM evaluation in patient education has remained incomplete for a clinically important reason: TAM alone cannot account for how clinicians&#x2019; attitudes toward a patient&#x2019;s medical condition shape their judgment of an LLM output&#x2019;s quality, relevance, and appropriateness.</p><p>Applied to LLM evaluation in patient education, TAM&#x2019;s constructs map productively onto the evaluation problem. PU corresponds to whether an LLM output contains accurate, complete, and guideline-adherent information that a clinician would trust. PEOU corresponds to whether the output is written in plain language that patients with varying health literacy levels can readily understand. These mappings are clinically meaningful (<xref ref-type="table" rid="table1">Table 1</xref>). However, they leave a critical evaluative gap unaddressed: when expert clinicians serve as proxy raters of patient-facing AI outputs, their judgments are filtered through attitudes, assumptions, and levels of regard toward the clinical conditions and patient populations being evaluated.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Mapping of Technology Acceptance Model (TAM) constructs to the 4 TAM-Medical Condition Regard Scale (MCRS) evaluation criteria.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">TAM construct</td><td align="left" valign="bottom">Definition</td><td align="left" valign="bottom">Evaluation criterion</td><td align="left" valign="bottom">Guiding evaluation question</td></tr></thead><tbody><tr><td align="left" valign="top">Perceived usefulness</td><td align="left" valign="top">The degree to which a person believes using a technology would enhance their performance</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top">Does the output contain guideline-adherent, actionable information a clinician would trust?</td></tr><tr><td align="left" valign="top">Perceived usefulness</td><td align="left" valign="top">The degree to which a person believes using a technology would enhance their performance</td><td align="left" valign="top">Completeness</td><td align="left" valign="top">Does the output cover all required clinical and social determinants of health (SDoH) elements a patient needs to act safely?</td></tr><tr><td align="left" valign="top">Perceived ease of use</td><td align="left" valign="top">The degree to which a person believes using a technology would be free of effort</td><td align="left" valign="top">Clarity</td><td align="left" valign="top">Is the output written in plain language at or below a 6th-grade reading level per Agency for Healthcare Research and Quality standards?</td></tr><tr><td align="left" valign="top">Perceived ease of use</td><td align="left" valign="top">The degree to which a person believes using a technology would be free of effort</td><td align="left" valign="top">Appropriateness</td><td align="left" valign="top">Does the output reflect the patient&#x2019;s specific SDoH context, language preference, and communication needs?</td></tr></tbody></table></table-wrap></sec><sec id="s2-2"><title>The MCRS: A Theoretically Grounded Lens for Attitudinal Variance</title><p>The MCRS, developed by Christison et al [<xref ref-type="bibr" rid="ref21">21</xref>], is a validated instrument that measures clinicians&#x2019; attitudes and regard toward patients with specific medical diagnoses. Originally applied in medical education to assess attitudinal confounds toward stigmatized conditions such as substance use disorders, obesity, and mental illness, the MCRS captures a dimension of clinical evaluation that technical rubrics systematically miss&#x2014;the attitudinal context in which clinician judgment operates.</p><p>In the context of LLM evaluation for patient education, MCRS-inspired constructs address a specific and underexamined validity threat. When an expert clinician panel rates LLM-generated postcare patient education for a patient with depression, opioid use disorder, or morbid obesity, their attitudinal regard for that condition may influence how they score the output&#x2019;s appropriateness, tone, and personalization, independent of the LLM&#x2019;s actual performance. Without accounting for this dimension, clinician panel ratings may reflect evaluator attitudinal variance as much as model quality, undermining the validity of the evaluation.</p></sec></sec><sec id="s3"><title>TAM-MCRS: A Combination Prescription</title><p>The TAM-MCRS LLM evaluation framework addresses a gap that existing evaluation approaches were not designed to close. Most published clinical LLM evaluations measure performance against clinician standards, asking whether a model&#x2019;s output is accurate enough for a clinician to trust. This framework reorients the evaluative question in two directions simultaneously: is the output accessible, personalized, and equitable enough for a patient to use, and how might measured clinician attitudinal variance influence the ratings used to answer that question?</p><p>The first reorientation reflects a core health informatics principle: technology adoption is not determined by technical performance alone, but by the PU and PEOU experienced by the people the technology is meant to serve [<xref ref-type="bibr" rid="ref18">18</xref>]. The second reorientation, introducing the MCRS as an attitudinal covariate, reflects the discipline&#x2019;s accountability to equity. Clinician panels are the gold standard for LLM evaluation in health care, but they are not attitudinally neutral. A framework that does not account for this difference is not fully valid for disparity-responsive research.</p><p>The stratified, disparity-responsive analysis built into this framework, which examines model performance across vignette race, language, social determinants of health (SDoH) tier, and age, positions equity not as an add-on but as a core dimension of evaluation. Prior research has documented that LLMs are less accurate for populations underrepresented in their training data [<xref ref-type="bibr" rid="ref22">22</xref>] and less factually precise for subjects that appear there infrequently [<xref ref-type="bibr" rid="ref23">23</xref>]. This framework applies that insight to patient demographics, testing whether performance gaps emerge systematically for underserved patient profiles. The MCRS covariate analysis adds a second disparity-responsive validity layer, ensuring that observed performance gaps reflect model differences rather than evaluator attitudinal variance.</p><p>Multiple frameworks for evaluating LLMs in health care have emerged in recent years, including comprehensive human-evaluation frameworks such as QUEST (Quality of Information, Understanding and Reasoning, Expression Style and Persona, Safety and Harm, and Trust and Confidence), which organizes clinician evaluation around quality of information, understanding and reasoning, expression style and persona, safety and harm, and trust and confidence [<xref ref-type="bibr" rid="ref24">24</xref>], and multidimensional benchmarking frameworks such as MEDIC (medical reasoning, ethics and bias, data and language understanding, in-context learning, and clinical safety), which assesses medical reasoning, ethics and bias, data and language understanding, in-context learning, and clinical safety [<xref ref-type="bibr" rid="ref25">25</xref>]. Recent surveys of LLM evaluation in health care similarly observe that prior evaluation has focused narrowly on accuracy and call for broader, multidimensional criteria spanning safety, reasoning, and clinical applicability [<xref ref-type="bibr" rid="ref26">26</xref>]. Separately, TAM is widely used to study clinician adoption of health technologies [<xref ref-type="bibr" rid="ref20">20</xref>], and the MCRS measures clinician attitudes toward specific patient conditions [<xref ref-type="bibr" rid="ref21">21</xref>]. To our knowledge, however, no existing framework integrates these 2 instruments as paired, measured constructs within a clinician-centered evaluation of LLM-generated patient education. Existing work occupies adjacent space, applying TAM to technology adoption or proposing standardized human-evaluation workflows for health care LLMs. However, none combines TAM and MCRS to model clinician usability perceptions and condition-specific regard as analytic covariates alongside output quality, readability, and understandability.</p></sec><sec id="s4"><title>Our TAM-MCRS LLM Evaluation Framework</title><sec id="s4-1"><title>Conceptual Architecture</title><p>We organized our framework around a 2-arm evaluator design that can be applied to the outputs of one or more models. In the clinician arm, an expert clinician panel evaluates each output with the TAM-MCRS instrument, which integrates 4-criterion quantitative scoring with the MCRS attitudinal regard measure, serving as the primary clinical reference for comparison. In the data science arm, automated methods score the same outputs independently. Automated scores are used for parallel assessment and discrepancy monitoring. When automated scores diverge from clinician ratings, the divergence should be reported descriptively rather than resolved through adjudication. The MCRS attitudinal covariate contextualizes the clinician ratings. Together, these components produce a replicable, multidimensional approach to assessing LLM-generated patient education that accounts for both output quality and the attitudinal context in which clinician evaluation operates (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><p>The MCRS-inspired survey layer contextualizes the quantitative ratings by surfacing attitudinal influences that may have shaped clinician judgment, providing both a validity check and a source of qualitative data for framework refinement (<xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Technology Acceptance Model-Medical Condition Regard Scale (TAM-MCRS) conceptual framework for clinician-centered large language model (LLM) evaluation in patient education. TAM constructs (perceived usefulness [PU] and perceived ease of use [PEOU]) map onto 4 evaluation criteria [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>]. The MCRS serves as the attitudinal covariate layer, supporting examination of measured evaluator attitudinal variance in clinician panel ratings [<xref ref-type="bibr" rid="ref21">21</xref>]. SDoH: social determinants of health.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e100423_fig01.png"/></fig><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Proposed Technology Acceptance Model-Medical Condition Regard Scale (TAM-MCRS) large language model (LLM) evaluation workflow. The framework begins with clinically realistic synthetic vignettes, generates patient education outputs from one or more models using a standardized master prompt template, and evaluates these outputs across 2 arms. The clinician arm administers the TAM-MCRS instrument, which combines 4-criterion quantitative scoring with MCRS attitudinal regard items. The data science arm applies established automated assessment methods across the same 4 criteria. Paired scores are then compared across arms, with the MCRS score used as an exploratory covariate.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e100423_fig02.png"/></fig></sec><sec id="s4-2"><title>The Clinician Arm</title><p>In the clinician arm, an expert panel works through a single structured TAM-MCRS instrument. The TAM component captures each output&#x2019;s PU and PEOU from a patient perspective; because these constructs operationalize the questions this framework asks, they serve as criterion dimensions rather than adoption predictors. The MCRS component is adapted as a postevaluation covariate, retaining the core domains of clinician regard. Clinicians rate their regard toward the diagnosis represented in the vignette, not toward the output itself, so that attitudinal regard can be examined as a possible source of rating variance independent of the model that produced it [<xref ref-type="bibr" rid="ref21">21</xref>]. Because the adapted instrument has not been independently validated in this configuration, its use is explicitly exploratory. The criterion mapping shown in <xref ref-type="table" rid="table1">Table 1</xref> governs which instrument items inform which score.</p></sec><sec id="s4-3"><title>The Data Science Arm</title><p>In the data science arm, outputs from each model are evaluated against the 4 primary TAM criteria, with each criterion matched to an evidence-based scoring method (<xref ref-type="table" rid="table2">Table 2</xref>). The 4 criteria and their scoring approaches were designed to address the specific limitations of existing LLM benchmarks, which rely primarily on n-gram statistical approaches such as BLEU (Bilingual Evaluation Understudy) [<xref ref-type="bibr" rid="ref27">27</xref>] and ROUGE (Recall-Oriented Understudy for Gisting Evaluation) [<xref ref-type="bibr" rid="ref28">28</xref>]; these approaches cannot capture semantic nuance, population-specific communication needs, or health literacy adaptation.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Four-criterion Technology Acceptance Model-Medical Condition Regard Scale (TAM-MCRS) evaluation matrix.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Criterion</td><td align="left" valign="bottom">Definition</td></tr></thead><tbody><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">Factual correctness of diagnoses, medications, red flags, and follow-up vs clinical guidelines</td></tr><tr><td align="left" valign="top">Appropriateness</td><td align="left" valign="top">Social determinants of health personalization, health literacy adaptation, and population-specific communication needs</td></tr><tr><td align="left" valign="top">Clarity</td><td align="left" valign="top">Plain language quality, reading level, and patient-facing tone</td></tr><tr><td align="left" valign="top">Completeness</td><td align="left" valign="top">Coverage of all required education elements per vignette condition</td></tr></tbody></table></table-wrap><p><xref ref-type="table" rid="table2">Table 2</xref> defines each criterion. Two features matter for the argument rather than the mechanics. Clarity carries an a priori directional hypothesis: a biomedical training corpus may elevate clinical vocabulary at the expense of accessibility, so a model that scores well on accuracy may score poorly on the dimension that decides whether a patient can act. Completeness is scored in both arms, which is what makes the interarm comparison possible. The content checklist is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> and an illustrative vignette in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. In future empirical applications, established automated methods such as FActScore [<xref ref-type="bibr" rid="ref23">23</xref>], G-Eval [<xref ref-type="bibr" rid="ref29">29</xref>], and Flesch-Kincaid readability with the Patient Education Materials Assessment Tool (PEMAT) [<xref ref-type="bibr" rid="ref30">30</xref>] may support these criteria alongside clinician judgment.</p><p>To make this concrete, consider how the framework would operate in practice. A health system is choosing between 2 language models to draft discharge instructions. The patient is an older adult with hypertension and type 2 diabetes, on multiple medications, facing transportation and cost barriers to filling prescriptions. Both models generate the patient&#x2019;s instructions from the same prompt. A clinician on the panel reads each version and scores it on the 4 criteria, without seeing which model produced it, then records their regard for the conditions in the vignette. Meanwhile, the automated pipeline scores the same 2 outputs on the same 4 criteria. The system now has 2 independent readings of each output, a measure with which to examine whether the rater&#x2019;s own attitude may have shaped their scores, and a defensible basis for choosing between the models.</p></sec></sec><sec id="s5"><title>Our Future Directions</title><p>This framework is designed to serve as the evaluative foundation for a subsequent Readability, Disparity-responsive design, and Personalization prompting framework and a future multisite R01 clinical trial examining AI-enabled postcare communication quality across safety-net health systems. When applying our framework in evaluations, criterion scores would be compared across model groups, with the MCRS score entered as an exploratory covariate to assess whether condition-specific clinician regard is associated with rating variation. Repeated ratings and the measurement properties of each criterion would be addressed using appropriate statistical methods, as detailed in the subsequent empirical report. Exploratory stratification would examine whether performance differences vary by vignette race or ethnicity, language preference, SDoH tier, and age group. Future applications of the framework, whether by our collective or others, should publicly document model versions, model prompts, scoring prompts, rubric specifications, and analytic assumptions before data collection to support reproducibility. Full analytic specification, including model syntax and prespecified hypotheses, is provided in the accompanying materials. We recognize that the primary limitation of the framework is that the PEOU scores are currently designed to be provided by clinician proxies rather than by patients themselves. In the future, we are dedicated to incorporating direct patient evaluators from the target populations, including those in the Louisiana Diabetes Belt and Cancer Alley, to cross-validate clinician proxy assessments of patient-facing accessibility against the health literacy and communication barriers experienced by the populations this framework is designed to serve.</p></sec><sec id="s6"><title>From Bedside to Boardrooms: Framework Impact</title><p>At the bedside, informaticists and clinical educators can use the 4-criterion rubric as a practical vetting tool before deploying any LLM-generated patient education material. The checklist-based completeness scoring and Flesch-Kincaid grade-level targets are straightforward to apply without specialized AI expertise, making the framework accessible to frontline staff in postcare communication workflows.</p><p>At the health system level, the framework provides a structured basis for LLM procurement and governance decisions. Health systems currently adopt LLMs with limited standardized evaluation criteria for patient-facing applications. A reproducible, multimethod evaluation framework gives informatics leaders a defensible, evidence-based process for comparing tools before deployment, with a documentation trail that supports institutional review board and regulatory accountability.</p><p>At the disciplinary level, this framework advances health informatics as a field that evaluates not only whether AI tools work but also whether they work responsibly and whether our methods for determining that are themselves free from attitudinal confounders. As LLMs become increasingly embedded in clinical workflows, informaticists are uniquely positioned to lead evaluation efforts that center patient experience, clinician accountability, and population health outcomes.</p></sec></body><back><ack><p>The authors thank the Artificial Intelligence/Machine Learning Consortium to Advance Health Equity and Researcher Diversity (AIM-AHEAD) Clinicians Leading Ingenuity IN AI Quality (CLINAQ) program leadership for fellowship support and mentorship throughout this work. The authors also acknowledge the Veterans Affairs Quality Scholars fellowship program for supporting the foundational conceptualization of this framework and Ochsner Health and the Xavier University of Louisiana Department of Public Health Sciences for ongoing institutional support.</p><p>The authors declare the use of generative AI in the writing process. According to the Generative AI Delegation Taxonomy (2025), the following tasks were delegated to generative AI tools under full human supervision: visualization, proofreading and editing, reformatting, and publication support. The generative AI tools used were ChatGPT (GPT-4o; OpenAI), Claude (Sonnet 4.6 and Opus 4.8; Anthropic), and the Gamma app. Responsibility for the final manuscript lies entirely with the authors. Generative AI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>The research reported in this paper was supported by the Artificial Intelligence/Machine Learning Consortium to Advance Health Equity and Researcher Diversity (AIM-AHEAD) Coordinating Center at the University of North Texas Health Science Center at Fort Worth. This research was, in part, funded by National Institutes of Health (NIH) agreement number 1OT2OD032581. The views and conclusions contained in this document are those of the authors and should not be interpreted as representing the official policies, either expressed or implied, of the NIH.</p></sec><sec><title>Data Availability</title><p>This is a framework paper. No empirical datasets are associated with this manuscript. To support transparent future use of the framework, the master prompt template, illustrative vignette structure, evaluation rubrics, and versioned scoring materials are intended to be made publicly available in nonproprietary formats, consistent with Artificial Intelligence/Machine Learning Consortium to Advance Health Equity and Researcher Diversity (AIM-AHEAD) open science and NIH Data Management and Sharing Policy commitments.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: DA, CP, RNA</p><p>Methodology: DA, GLW, CG, MOA, MM</p><p>Visualization: DA</p><p>Writing&#x2014;original draft: DA</p><p>Writing&#x2014;review and editing: DA, GLW, CG, MOA, MM, CP, RNA, JBH</p><p>All authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AIM-AHEAD</term><def><p>Artificial Intelligence/Machine Learning Consortium to Advance Health Equity and Researcher Diversity</p></def></def-item><def-item><term id="abb2">BLEU</term><def><p>Bilingual Evaluation Understudy</p></def></def-item><def-item><term id="abb3">CLINAQ</term><def><p>Clinicians Leading Ingenuity IN AI Quality</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">MCRS</term><def><p>Medical Condition Regard Scale</p></def></def-item><def-item><term id="abb6">MEDIC</term><def><p>medical reasoning, ethics and bias, data and language understanding, in-context learning, and clinical safety</p></def></def-item><def-item><term id="abb7">PEMAT</term><def><p>Patient Education Materials Assessment Tool</p></def></def-item><def-item><term id="abb8">PEOU</term><def><p>perceived ease of use</p></def></def-item><def-item><term id="abb9">PU</term><def><p>perceived usefulness</p></def></def-item><def-item><term id="abb10">QUEST</term><def><p>Quality of Information, Understanding and Reasoning, Expression Style and Persona, Safety and Harm, and Trust and Confidence</p></def></def-item><def-item><term id="abb11">ROUGE</term><def><p>Recall-Oriented Understudy for Gisting Evaluation</p></def></def-item><def-item><term id="abb12">SDoH</term><def><p>social determinants of health</p></def></def-item><def-item><term id="abb13">TAM</term><def><p>Technology Acceptance Model</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Weiss</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>HJ</given-names> </name></person-group><article-title>Overview of clinical conditions with frequent and costly hospital readmissions by payer, 2018</article-title><source>Healthcare Cost and Utilization Project (HCUP) Statistical Briefs</source><year>2021</year><publisher-name>Agency for Healthcare Research and Quality (US)</publisher-name></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Galarraga</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Pines</surname><given-names>JM</given-names> </name></person-group><article-title>Costs of ED episodes of care in the United States</article-title><source>Am J Emerg Med</source><year>2016</year><month>03</month><volume>34</volume><issue>3</issue><fpage>357</fpage><lpage>365</lpage><pub-id pub-id-type="doi">10.1016/j.ajem.2015.06.001</pub-id><pub-id pub-id-type="medline">26763823</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>DeSai</surname><given-names>C</given-names> </name><name name-style="western"><surname>Janowiak</surname><given-names>K</given-names> </name><name name-style="western"><surname>Secheli</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Empowering patients: simplifying discharge instructions</article-title><source>BMJ Open Qual</source><year>2021</year><month>09</month><volume>10</volume><issue>3</issue><fpage>e001419</fpage><pub-id pub-id-type="doi">10.1136/bmjoq-2021-001419</pub-id><pub-id pub-id-type="medline">34521621</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Karnan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Francis</surname><given-names>J</given-names> </name><name name-style="western"><surname>Vijayvargiya</surname><given-names>I</given-names> </name><name name-style="western"><surname>Rubino Tan</surname><given-names>C</given-names> </name></person-group><article-title>Analyzing the effectiveness of AI-generated patient education materials: a comparative study of ChatGPT and Google Gemini</article-title><source>Cureus</source><year>2024</year><month>11</month><volume>16</volume><issue>11</issue><fpage>e74398</fpage><pub-id pub-id-type="doi">10.7759/cureus.74398</pub-id><pub-id pub-id-type="medline">39723279</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shaari</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Bhalla</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Comparative analysis of artificial intelligence platforms in generating post-operative instructions for rhinologic surgery</article-title><source>Indian J Otolaryngol Head Neck Surg</source><year>2025</year><month>01</month><volume>77</volume><issue>1</issue><fpage>601</fpage><lpage>603</lpage><pub-id pub-id-type="doi">10.1007/s12070-024-05161-1</pub-id><pub-id pub-id-type="medline">40066407</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Swisher</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>AW</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>GC</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Carle</surname><given-names>TR</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>DM</given-names> </name></person-group><article-title>Enhancing health literacy: evaluating the readability of patient handouts revised by ChatGPT&#x2019;s large language model</article-title><source>Otolaryngol Head Neck Surg</source><year>2024</year><month>12</month><volume>171</volume><issue>6</issue><fpage>1751</fpage><lpage>1757</lpage><pub-id pub-id-type="doi">10.1002/ohn.927</pub-id><pub-id pub-id-type="medline">39105460</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rooney</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Sachdev</surname><given-names>S</given-names> </name><name name-style="western"><surname>Byun</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jagsi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Golden</surname><given-names>DW</given-names> </name></person-group><article-title>Readability of patient education materials in radiation oncology-are we improving?</article-title><source>Pract Radiat Oncol</source><year>2019</year><month>11</month><volume>9</volume><issue>6</issue><fpage>435</fpage><lpage>440</lpage><pub-id pub-id-type="doi">10.1016/j.prro.2019.06.005</pub-id><pub-id pub-id-type="medline">31228657</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bindhu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Nattam</surname><given-names>A</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Roles of health literacy in relation to social determinants of health and recommendations for informatics-based interventions: systematic review</article-title><source>Online J Public Health Inform</source><year>2024</year><month>03</month><day>20</day><volume>16</volume><fpage>e50898</fpage><pub-id pub-id-type="doi">10.2196/50898</pub-id><pub-id pub-id-type="medline">38506914</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bender</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Gebru</surname><given-names>T</given-names> </name><name name-style="western"><surname>McMillan-Major</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shmitchell</surname><given-names>S</given-names> </name></person-group><article-title>On the dangers of stochastic parrots: can language models be too big?</article-title><source>Proceedings of the 2021 ACM Conference on Fairness, Accountability, and Transparency</source><year>2021</year><publisher-name>Association for Computing Machinery</publisher-name><pub-id pub-id-type="doi">10.1145/3442188.3445922</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hoffmann</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rueger</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Current applications and challenges in large language models for patient care: a systematic review</article-title><source>Commun Med (Lond)</source><year>2025</year><month>01</month><day>21</day><volume>5</volume><issue>1</issue><fpage>26</fpage><pub-id pub-id-type="doi">10.1038/s43856-024-00717-2</pub-id><pub-id pub-id-type="medline">39838160</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Norori</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Aellen</surname><given-names>FM</given-names> </name><name name-style="western"><surname>Faraci</surname><given-names>FD</given-names> </name><name name-style="western"><surname>Tzovara</surname><given-names>A</given-names> </name></person-group><article-title>Addressing bias in big data and AI for health care: a call for open science</article-title><source>Patterns (N Y)</source><year>2021</year><month>10</month><day>8</day><volume>2</volume><issue>10</issue><fpage>100347</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2021.100347</pub-id><pub-id pub-id-type="medline">34693373</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Panch</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mattie</surname><given-names>H</given-names> </name><name name-style="western"><surname>Atun</surname><given-names>R</given-names> </name></person-group><article-title>Artificial intelligence and algorithmic bias: implications for health systems</article-title><source>J Glob Health</source><year>2019</year><month>12</month><volume>9</volume><issue>2</issue><fpage>010318</fpage><pub-id pub-id-type="doi">10.7189/jogh.09.020318</pub-id><pub-id pub-id-type="medline">31788229</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barker</surname><given-names>LE</given-names> </name><name name-style="western"><surname>Kirtland</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Gregg</surname><given-names>EW</given-names> </name><name name-style="western"><surname>Geiss</surname><given-names>LS</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>TJ</given-names> </name></person-group><article-title>Geographic distribution of diagnosed diabetes in the U.S.: a diabetes belt</article-title><source>Am J Prev Med</source><year>2011</year><month>04</month><volume>40</volume><issue>4</issue><fpage>434</fpage><lpage>439</lpage><pub-id pub-id-type="doi">10.1016/j.amepre.2010.12.019</pub-id><pub-id pub-id-type="medline">21406277</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="report"><article-title>National diabetes statistics report</article-title><year>2026</year><access-date>2026-09-02</access-date><publisher-name>U.S. Centers for Disease Control and Prevention</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/diabetes/php/data-research/index.html">https://www.cdc.gov/diabetes/php/data-research/index.html</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>United States cancer statistics: data visualizations</article-title><source>U.S. Centers for Disease Control and Prevention</source><access-date>2026-05-06</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://gis.cdc.gov/Cancer/USCS/#/">https://gis.cdc.gov/Cancer/USCS/#/</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><article-title>Cancer incidence in Louisiana by census tract - 2023</article-title><source>LSU Health New Orleans</source><year>2023</year><access-date>2026-05-06</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://sph.lsuhsc.edu/louisiana-tumor-registry/data-usestatistics/monographs-publications/cancer-incidence-in-louisiana-by-census-tract-2023/">https://sph.lsuhsc.edu/louisiana-tumor-registry/data-usestatistics/monographs-publications/cancer-incidence-in-louisiana-by-census-tract-2023/</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alam</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Williams</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kibriya</surname><given-names>MG</given-names> </name><etal/></person-group><article-title>Association between diagnostic history and cancer incidence within 5 years: a real-world observational analysis</article-title><source>Cancer Res Commun</source><year>2026</year><month>05</month><day>1</day><volume>6</volume><issue>5</issue><fpage>1083</fpage><lpage>1091</lpage><pub-id pub-id-type="doi">10.1158/2767-9764.CRC-26-0163</pub-id><pub-id pub-id-type="medline">41996637</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Davis</surname><given-names>FD</given-names> </name></person-group><article-title>Perceived usefulness, perceived ease of use, and user acceptance of information technology</article-title><source>MIS Q</source><year>1989</year><month>09</month><volume>13</volume><issue>3</issue><fpage>319</fpage><lpage>340</lpage><pub-id pub-id-type="doi">10.2307/249008</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Venkatesh</surname><given-names>V</given-names> </name><name name-style="western"><surname>Morris</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Davis</surname><given-names>GB</given-names> </name><name name-style="western"><surname>Davis</surname><given-names>FD</given-names> </name></person-group><article-title>User acceptance of information technology: toward a unified view</article-title><source>MIS Q</source><year>2003</year><month>09</month><day>1</day><volume>27</volume><issue>3</issue><fpage>425</fpage><lpage>478</lpage><pub-id pub-id-type="doi">10.2307/30036540</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tetik</surname><given-names>G</given-names> </name><name name-style="western"><surname>T&#x00FC;rkeli</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pinar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tarim</surname><given-names>M</given-names> </name></person-group><article-title>Health information systems with technology acceptance model approach: a systematic review</article-title><source>Int J Med Inform</source><year>2024</year><month>10</month><volume>190</volume><fpage>105556</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2024.105556</pub-id><pub-id pub-id-type="medline">39053345</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Christison</surname><given-names>GW</given-names> </name><name name-style="western"><surname>Haviland</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Riggs</surname><given-names>ML</given-names> </name></person-group><article-title>The medical condition regard scale: measuring reactions to diagnoses</article-title><source>Acad Med</source><year>2002</year><month>03</month><volume>77</volume><issue>3</issue><fpage>257</fpage><lpage>262</lpage><pub-id pub-id-type="doi">10.1097/00001888-200203000-00017</pub-id><pub-id pub-id-type="medline">11891166</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maity</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saikia</surname><given-names>MJ</given-names> </name></person-group><article-title>Large language models in healthcare and medical applications: a review</article-title><source>Bioengineering (Basel)</source><year>2025</year><month>06</month><day>10</day><volume>12</volume><issue>6</issue><fpage>631</fpage><pub-id pub-id-type="doi">10.3390/bioengineering12060631</pub-id><pub-id pub-id-type="medline">40564447</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Min</surname><given-names>S</given-names> </name><name name-style="western"><surname>Krishna</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lyu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>FActScore: fine-grained atomic evaluation of factual precision in long form text generation</article-title><source>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>12076</fpage><lpage>12100</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.741</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tam</surname><given-names>TY</given-names> </name><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kapoor</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>28</day><volume>7</volume><issue>1</issue><fpage>258</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id><pub-id pub-id-type="medline">39333376</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kanithi</surname><given-names>PK</given-names> </name><name name-style="western"><surname>Christophe</surname><given-names>C</given-names> </name><name name-style="western"><surname>Pimentel</surname><given-names>MA</given-names> </name><etal/></person-group><article-title>MEDIC: towards a comprehensive framework for evaluating LLMs in clinical applications</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 11, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2409.07314</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>He</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>D</given-names> </name></person-group><article-title>Evaluating large language models and agents in healthcare: key challenges in clinical applications</article-title><source>Intell Med</source><year>2025</year><month>05</month><volume>5</volume><issue>2</issue><fpage>151</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.imed.2025.03.002</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Papineni</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roukos</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>WJ</given-names> </name></person-group><article-title>BLEU: a method for automatic evaluation of machine translation</article-title><source>ACL &#x2019;02: Proceedings of the 40th Annual Meeting on Association for Computational Linguistics</source><year>2002</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>311</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.3115/1073083.1073135</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name></person-group><article-title>ROUGE: a package for automatic evaluation of summaries</article-title><source>Text Summarization Branches Out</source><year>2004</year><access-date>2026-09-02</access-date><publisher-name>Association for Computational Linguistics</publisher-name><fpage>74</fpage><lpage>81</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W04-1013/">https://aclanthology.org/W04-1013/</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Iter</surname><given-names>D</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>C</given-names> </name></person-group><article-title>G-Eval: NLG evaluation using GPT-4 with better human alignment</article-title><source>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>2511</fpage><lpage>2522</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.153</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shoemaker</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Wolf</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Brach</surname><given-names>C</given-names> </name></person-group><article-title>Development of the Patient Education Materials Assessment Tool (PEMAT): a new measure of understandability and actionability for print and audiovisual patient information</article-title><source>Patient Educ Couns</source><year>2014</year><month>09</month><volume>96</volume><issue>3</issue><fpage>395</fpage><lpage>403</lpage><pub-id pub-id-type="doi">10.1016/j.pec.2014.05.027</pub-id><pub-id pub-id-type="medline">24973195</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Proposed master prompt template. The standardized prompt template we propose is to be applied uniformly across all models evaluated at temperature 0. It includes a Flesch-Kincaid grade-level target, 7-element content checklist, social determinants of health context injection fields, patient-centered framing directives, section headers for completeness scoring, dynamic length guidance for reusable patient education workflows, and mandatory safety element requirements. For the Clinicians Leading Ingenuity IN AI Quality (CLINAQ) application in which we propose to use it, the target is the multimorbidity and polypharmacy pathway at 500 to 650 words.</p><media xlink:href="formative_v10i1e100423_app1.docx" xlink:title="DOCX File, 39 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Illustrative sample clinical vignette. A representative synthetic, deidentified clinical vignette (hypertension with type 2 diabetes comorbidity) is provided to illustrate the proposed evaluation workflow, reflecting the multimorbidity, polypharmacy, and social determinants of health (SDoH) complexity the framework is designed to address. It includes patient profile, medication list, SDoH fields, and evaluation scoring guidance.</p><media xlink:href="formative_v10i1e100423_app2.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material></app-group></back></article>