<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e86906</article-id><article-id pub-id-type="doi">10.2196/86906</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Measuring Depression Severity With Clinical Global Impression&#x2013;Severity Scale Scores From Clinical Notes Using Large Language Models: Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Kevin</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zirikly</surname><given-names>Ayah</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Collica</surname><given-names>Sarah C</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Goes</surname><given-names>Fernando S</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Congwen</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Nguyen</surname><given-names>Trang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gagliardi</surname><given-names>Jane P</given-names></name><degrees>MD, MHS</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Goldstein</surname><given-names>Benjamin A</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hong</surname><given-names>Hwanhee</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Stuart</surname><given-names>Elizabeth A</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Zandi</surname><given-names>Peter P</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Psychiatry and Behavioral Sciences, School of Medicine, Johns Hopkins University</institution><addr-line>550 N Broadway</addr-line><addr-line>Baltimore</addr-line><addr-line>MD</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Computer Science, Whiting School of Engineering, Johns Hopkins University</institution><addr-line>Baltimore</addr-line><addr-line>MD</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Computer Science, School of Engineering and Applied Science, George Washington University</institution><addr-line>Washington</addr-line><addr-line>DC</addr-line><country>United States</country></aff><aff id="aff4"><institution>Department of Biostatistics and Bioinformatics, School of Medicine, Duke University</institution><addr-line>Durham</addr-line><addr-line>NC</addr-line><country>United States</country></aff><aff id="aff5"><institution>Department of Biostatistics, Bloomberg School of Public Health, Johns Hopkins University</institution><addr-line>Baltimore</addr-line><addr-line>MD</addr-line><country>United States</country></aff><aff id="aff6"><institution>Department of Psychiatry and Behavioral Sciences, School of Medicine, Duke University</institution><addr-line>Durham</addr-line><addr-line>NC</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Mavragani</surname><given-names>Amaryllis</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Shin</surname><given-names>Daun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yoshimura</surname><given-names>Kenshuke</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Peter P Zandi, PhD, Department of Psychiatry and Behavioral Sciences, School of Medicine, Johns Hopkins University, 550 N Broadway, Baltimore, MD, 21205, United States, 1 410-614-1923; <email>pzandi1@jhu.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>8</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e86906</elocation-id><history><date date-type="received"><day>02</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>22</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>27</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Kevin Li, Ayah Zirikly, Sarah C Collica, Fernando S Goes, Congwen Zhao, Trang Nguyen, Jane P Gagliardi, Benjamin A Goldstein, Hwanhee Hong, Elizabeth A Stuart, Peter P Zandi. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 10.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e86906"/><abstract><sec><title>Background</title><p>Real-world psychiatric care is marked by wide heterogeneity in clinical presentations and outcomes, underscoring the need for systematic approaches to outcome measurement. The Clinical Global Impression&#x2013;Severity (CGI-S) scale is a brief, clinician-rated measure of overall illness severity that is widely used in psychiatric research, but rarely documented in routine care. Large language models (LLMs) may enable automated extraction of CGI-S scores from narrative clinical notes, thereby providing scalable outcome measures for real-world clinical care and research.</p></sec><sec><title>Objective</title><p>The study aimed to evaluate whether LLMs can estimate CGI-S scores from psychiatric clinical notes for patients with major depressive disorder (MDD) and to compare performance across prompting strategies and model architectures.</p></sec><sec sec-type="methods"><title>Methods</title><p>We extracted psychiatrist-authored notes from the Johns Hopkins electronic health record. Three board-certified psychiatrists independently rated 77 clinical notes using a validated depression-specific Clinical Global Impression (CGI) rubric. Weighted Cohen kappa coefficients were calculated to assess inter-rater reliability and model-human agreement. We evaluated GPT-4o under zero-shot and few-shot prompting conditions and Llama-4 under zero-shot prompting. Model performance was assessed by comparing LLM-generated scores to individual rater scores and consensus ratings. Exploratory analyses evaluated whether agreement varied by patient demographics, care setting, note length, or the percentage of copy-forwarded text within each note.</p></sec><sec sec-type="results"><title>Results</title><p>Interrater reliability among psychiatrists was high (&#x03BA;=0.77&#x2010;0.78). GPT-4o with zero-shot prompting demonstrated the highest agreement with average human ratings (&#x03BA;=0.85, 95% CI 0.78-0.90), and few-shot prompting did not improve performance. In contrast, Llama-4 with zero-shot prompting demonstrated lower agreement with average human ratings (&#x03BA;=0.70, 95% CI 0.55&#x2010;0.80). Model agreement did not significantly differ across age, sex, race, treatment location, or the percentage of copy-forwarded text, but it was significantly lower for notes below the median note length than for notes at or above the median length (&#x03BA;=0.72 vs 0.92; <italic>P</italic>=.003).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>LLMs can estimate clinician-rated CGI-S scores from psychiatric clinical notes for patients with MDD at a level of agreement comparable to that of expert interrater reliability. Performance varied by model architecture, with GPT-4o outperforming an open-source alternative. If further validated, this approach may support scalable outcome measurement in research settings and inform future efforts to implement measurement-based care in real-world psychiatric practice.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>natural language processing</kwd><kwd>psychiatry</kwd><kwd>depression</kwd><kwd>clinical global impression</kwd><kwd>clinical global impression&#x2013;severity</kwd><kwd>CGI-S</kwd><kwd>measurement-based care</kwd><kwd>electronic health records</kwd><kwd>interrater reliability</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Major depressive disorder (MDD) is a leading cause of disability worldwide, accounting for approximately one-third of years lived with disability from mental disorders, and its incidence has increased by approximately 60% between 1990 and 2019 [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. The clinical course and treatment response in MDD are highly variable, and many individuals experience incomplete remission or treatment resistance [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. Capturing this variability in real-world settings is important to gain a more nuanced phenotypic understanding of depression to inform more personalized care. Although clinical trials routinely use validated symptom rating scales, outcome data in routine psychiatric care are inconsistently measured [<xref ref-type="bibr" rid="ref5">5</xref>], with less than 20% of mental health providers incorporating measurement-based care (MBC) into their practice and as few as 5% using it consistently at every session [<xref ref-type="bibr" rid="ref6">6</xref>]. Although MBC has been shown to improve outcomes and is recommended in treatment guidelines [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>], real-world uptake has been slow due to practical and organizational obstacles [<xref ref-type="bibr" rid="ref9">9</xref>]. Implementation programs such as the National Network of Depression Centers Mood Outcomes Program have only recently demonstrated that adoption is feasible at scale when supported by coordinated infrastructure and system-level investment [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>In this context, the Clinical Global Impression (CGI) scale [<xref ref-type="bibr" rid="ref11">11</xref>] represents a promising candidate for scalable, measurement-based assessment in real-world settings. The CGI is a brief, Likert-type scale rated by clinicians during patient encounters and includes both the CGI-Severity (CGI-S) scale, which rates cross-sectional illness severity from 1 (not at all ill) to 7 (among the most extremely ill patients), and the CGI-Improvement (CGI-I) scale, which assesses longitudinal change from 1 (very much improved since initiation of treatment) to 7 (very much worse since initiation of treatment). The CGI-S scale, the focus of the present work and hereafter referred to simply as CGI, has been validated in a wide range of psychiatric settings (eg, inpatient [<xref ref-type="bibr" rid="ref12">12</xref>], outpatient [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref13">13</xref>], and clinical trial contexts [<xref ref-type="bibr" rid="ref14">14</xref>]) and psychiatric conditions including mood [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref18">18</xref>], anxiety [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>], and psychotic disorders [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. In inpatient care, CGI scores have shown strong convergent validity with established measures such as the Health of the Nation Outcome Scales (HoNOS), Mental Health Questionnaire&#x2013;14 (MHQ-14), and Depression Anxiety Stress Scales&#x2013;21 (DASS-21) [<xref ref-type="bibr" rid="ref12">12</xref>], underscoring their value as a brief and pragmatic outcome measure. In depression specifically, CGI ratings have demonstrated strong correlations with standard symptom scales (r&#x2248;0.6&#x2010;0.8), including the Beck Depression Inventory (BDI), Hamilton Depression Rating Scale (HAM-D), and Montgomery-&#x00C5;sberg Depression Rating Scale (MADRS), and have shown comparable or greater sensitivity to treatment-related change across outpatient and clinical trial populations [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>].</p><p>Interrater reliability for CGI has generally been reported to be in the moderate-to-excellent range across psychiatric disorders, with Cohen &#x03BA; or interrater coefficient values of approximately 0.7-0.8 [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref29">29</xref>]. However, critiques have highlighted its relative lack of specificity and ambiguous response anchors, which may contribute to variability in ratings [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>]. To address these concerns, Kadouri et al [<xref ref-type="bibr" rid="ref27">27</xref>] developed an improved CGI (iCGI) for depression, incorporating standardized case vignettes to guide rater calibration. This approach achieved excellent interrater reliability when averaging rater scores (intraclass correlation coefficient [ICC]&#x2248;0.9) and demonstrated greater sensitivity for detecting clinical change than with the HAM-D [<xref ref-type="bibr" rid="ref27">27</xref>].</p><p>Despite extensive validation in research and its potential for broad use across conditions, CGI scores are not routinely used or documented in clinical settings, where they could serve as a readily available outcome measure for large-scale observational studies. Instead, in typical clinical practice, details regarding symptom severity and improvement are recorded in the narrative text of clinical notes rather than quantified in a standardized way.</p><p>Advances in natural language processing (NLP) have introduced the possibility of extracting outcome measures directly from unstructured text. Previous approaches demonstrated that mood states and longitudinal treatment response could be accurately classified from outpatient psychiatric notes using rule-based NLP models [<xref ref-type="bibr" rid="ref33">33</xref>]. More broadly, NLP methods have been shown to consistently improve the performance of predictive models created solely from structured data [<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref38">38</xref>]. More recently, large language models (LLMs) have enabled the analysis of clinical text with a deeper understanding of linguistic context and meaning, moving beyond the keyword- or rule-based features used in previous NLP approaches [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>], with robust performance in both psychiatric feature extraction and classification tasks [<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref44">44</xref>]. However, recent work using LLMs (GPT-4o; OpenAI) to estimate depression severity scores from primary care notes demonstrated only modest performance, likely reflecting the limited psychiatric clinical documentation in primary care settings [<xref ref-type="bibr" rid="ref45">45</xref>].</p><p>Recent applications within psychiatry, where notes typically contain richer descriptions of symptoms and clinical impressions, have provided encouraging evidence that LLMs can generate structured clinical outcomes from narrative text. Wiest et al [<xref ref-type="bibr" rid="ref46">46</xref>] applied a Llama-2 model to psychiatric admission notes to classify the presence of suicidal ideation, plans, or attempts, achieving more than 85% accuracy and strong agreement with expert consensus ratings. McCoy and Perlis [<xref ref-type="bibr" rid="ref47">47</xref>] extended this work by applying reasoning-style LLM prompts to hospital discharge summaries to estimate suicide risk, yielding model-derived hazard ratios that were predictive of subsequent suicide or accidental death, while also generating step-by-step, interpretable textual explanations for the predictions. In a related study, GPT-4 was used to translate psychiatric emergency department notes into quantitative symptom ratings across domains such as mood and cognition and demonstrated that these scores could predict hospitalization and length of stay [<xref ref-type="bibr" rid="ref48">48</xref>].</p><p>Building on this foundation, we developed and tested a pipeline for using LLMs to generate CGI scores from the clinical notes of patients with depression. We first created a psychiatrist-annotated consensus reference and then compared different zero-shot and few-shot LLM prompting approaches to assess which showed the best alignment with clinician assessment. Through this approach, we sought to evaluate the feasibility of using LLMs to generate standardized outcome scores from unstructured psychiatric documentation, thereby bridging the gap between clinical assessment and scalable measurement for clinical use and research.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Data</title><p>This study was conducted using data from the Johns Hopkins Precision Medicine Center of Excellence in Mood Disorders registry, which was established to support large-scale clinical phenotyping and research in mood disorders. The registry includes longitudinal data from Epic (Epic Systems Corp), the electronic health record (EHR) of the Johns Hopkins Health System, on all patients with at least 1 encounter for a mood disorder (<italic>International Statistical Classification of Diseases and Related Health Problems, 10th revision</italic> [<italic>ICD-10</italic>]: F30.x, F31.x, F32.x, F33.x, F34.x, or F39*) who have been seen in the system since 2013. This was a retrospective study using routinely collected clinical data; no participants were prospectively recruited. For the current study, we extracted and analyzed a stratified random sample of clinical notes using the procedures described below.</p></sec><sec id="s2-2"><title>Clinical Note Selection and Processing</title><p>We extracted from the registry a stratified random sample of 499 psychiatrist-authored clinical notes from encounters with patients who had a primary diagnosis of MDD (<italic>ICD-10</italic> F32.x or F33.x) and were seen in the Department of Psychiatry and Behavioral Sciences at the Johns Hopkins Hospital (JHH) or Johns Hopkins Bayview Medical Center (JHBMC) between July 1, 2016 (when Epic implementation was completed enterprise-wide), and July 31, 2024. To capture a range of clinical severity, we randomly selected clinical notes in nearly equal proportion by sex and across 4 psychiatric care settings: inpatient, outpatient, intensive outpatient or partial day hospital, and consultation services. Each clinical note corresponded to a single clinical encounter, and for inpatient encounters for which multiple psychiatrist-authored progress notes were available, up to 2 notes per encounter were retained. The clinical notes reflected routine documentation generated at the time of care and included progress notes, consultation notes, or admission history and physical (H&#x0026;P) notes, depending on the clinical setting. Notes were authored by different psychiatrists practicing across these settings and reflected the variability in documentation style encountered in real-world psychiatric care. No minimum or maximum note length criteria were imposed, and no additional manual editing or segmentation of the note text was performed. Consequently, the clinical notes could contain copy-forwarded text or templated elements as would be seen in real-world practice.</p><p>Notes were randomly sampled from the full pool of 499 extracted notes to construct 2 nonoverlapping datasets: an initial calibration set (n=70) used for rater alignment and rubric calibration, and an independent test set (N=77) used for all primary analyses. The remaining notes were not used in the present study. All clinical notes were deidentified prior to annotation and model evaluation using an automated pipeline implemented in R, leveraging the <italic>spacyr</italic> package for named entity recognition along with custom preprocessing scripts to remove protected health information (PHI). Human raters and language models were provided with identical deidentified note text.</p></sec><sec id="s2-3"><title>Annotation and Rater Procedures</title><p>To establish a psychiatrist consensus reference set of annotated clinical notes for model calibration and testing, 3 board-certified psychiatrists (KL, FSG, and SCC) met to review the depression-validated CGI rubric developed by Kadouri et al [<xref ref-type="bibr" rid="ref27">27</xref>]. The 3 raters were study coauthors and were not blinded to the study objective, but all ratings were completed independently and before comparison with model outputs. Specifically, each rater independently assigned a CGI score to an initial set of notes (n=70) randomly drawn from the full dataset to calibrate the annotation procedure. The raters subsequently met to review notes with a deviation of &#x2265;2 points between any 2 raters (10/70, 14.3% notes) and resolved discrepancies through consensus discussion. We then randomly selected a new independent set of notes (N=77) to form the test dataset. Using the <italic>kappaSize</italic> package in R (version 1.2; R Foundation for Statistical Computing) [<xref ref-type="bibr" rid="ref49">49</xref>], we estimated that the test dataset would allow us to detect a Cohen &#x03BA; of 0.8 with a precision of at least &#x00B1;0.1 (95% CI width 0.2), assuming 3 raters and 7 ordinal CGI categories.</p></sec><sec id="s2-4"><title>Model Prompting and Output Evaluation</title><p>We developed a structured prompt for CGI scoring based on the same depression-validated CGI rubric [<xref ref-type="bibr" rid="ref27">27</xref>], which was reviewed by the human raters (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). We applied this prompt to each clinical note in the calibration dataset using OpenAI&#x2019;s GPT-4o (model identifier gpt-4o-2024-11-20). During the initial round of calibration, we observed that some clinical notes contained temporal dynamics (eg, &#x201C;depressed for several weeks but now improving&#x201D;) that were not adequately captured by the original phrasing. Because CGI-S is intended to reflect the clinician&#x2019;s global impression of severity at the time of the encounter rather than over a retrospective time frame, we refined the prompt to instruct the model to prioritize the patient&#x2019;s current clinical presentation. This revised version was used for all subsequent analyses with OpenAI&#x2019;s GPT-4o model on the test dataset of annotated clinical notes. We evaluated model performance under two conditions: (1) zero-shot, in which the model received only the prompt and the clinical note, and (2) few-shot, in which we added 6 example notes to the prompt, 1 for each CGI score from 1 to 6, selected from the initial calibration set of 70 notes for which all 3 human raters independently agreed on the score (no note was rated as 7). No model fine-tuning or supervised training was performed. To assess cross-model generalizability, we additionally evaluated an open-source LLM, Llama (version 4; Meta Platforms Inc) Maverick, (accessed through the Azure Databricks Foundation Model API end point databricks-llama-4-maverick, corresponding to the foundation model identifier system.ai.llama-4-maverick), on the independent test dataset using the same finalized zero-shot prompt and identical deidentified clinical notes used for the GPT-4o models.</p></sec><sec id="s2-5"><title>Statistical Analysis</title><p>We first assessed interrater reliability among the 3 human raters by calculating weighted Cohen kappa coefficients for each pair of raters (KL-FSG, KL-SCC, and FSG-SCC) using the independent test set of notes, where the weighted &#x03BA; statistic extends the original binary version to ordinal scales by accounting for the degree of disagreement between ratings. Conventionally, &#x03BA; values between 0.6 and 0.8 are interpreted as indicating substantial agreement, and values above 0.8 indicate excellent agreement [<xref ref-type="bibr" rid="ref50">50</xref>]. We then calculated pairwise weighted &#x03BA; values comparing the CGI scores generated by GPT-4o (zero-shot and few-shot prompting) and Llama-4 (zero-shot prompting) with those assigned by each human rater individually. CIs for weighted &#x03BA; values were estimated using nonparametric bootstrap resampling at the note level (5000 iterations).</p><p>To evaluate how well the LLM outputs aligned with consensus ratings from the human raters, we also compared model-generated CGI scores to reference scores defined in three ways: (1) the average score of all ratings rounded up as needed; (2) the unanimous score for notes where all 3 human raters independently assigned the same score; and (3) the majority score for notes where at least 2 of the 3 raters agreed.</p><p>Finally, we conducted exploratory analyses to assess whether the agreement between model-based scores and human ratings, as estimated by the weighted &#x03BA; values, was significantly different according to the age, sex, or race of the patient; the clinical location of the encounter; the length of the clinical note; or the percentage of copy-forwarded text in the note. For comparisons by clinical location, we counted intensive outpatient or partial day hospital encounters as inpatient and consultation services encounters as outpatient. For note length, we dichotomized notes by the median word count (1007, IQR 646-1535 words). For the percentage of copy-forwarded text, we dichotomized notes by the median percentage (81.3%, IQR 65.4%-88.5%). For all comparisons, we used average scores from all 3 human raters and the results from the GPT-4o zero-shot model, which demonstrated the highest agreement with human ratings. We evaluated the significance of differences in weighted &#x03BA; values between strata using a permutation procedure. Specifically, we randomly split the clinical notes into subsamples equal in size to the stratum of interest 5000 times and recalculated weighted &#x03BA; values within these randomly generated substrata. The proportion of iterations in which the difference in weighted &#x03BA; values was equal to or greater than the observed difference was taken as the empirical <italic>P</italic> value, representing the probability of obtaining the observed difference by chance. All analyses were carried out with the <italic>irr</italic> package in R [<xref ref-type="bibr" rid="ref51">51</xref>].</p></sec><sec id="s2-6"><title>Ethical Considerations</title><p>This study was approved by the Johns Hopkins School of Medicine institutional review board (IRB; IRB00312852) under a waiver of consent. All analyses of the clinical notes were conducted in a HIPAA (Health Insurance Portability and Accountability Act)-compliant computational environment following Johns Hopkins institutional policies. All analyses with OpenAI&#x2019;s GPT-4o model were carried out via a HIPAA-secure API with content logging and filtering disabled.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Descriptive Statistics</title><p><xref ref-type="table" rid="table1">Table 1</xref> shows the demographic characteristics of the patients whose clinical notes were included in our test dataset, along with the treatment locations from which the notes were drawn.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Test set patient demographic characteristics (N=77).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristics</td><td align="left" valign="bottom">Test set, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Age (years)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x003C;40</td><td align="left" valign="top">43 (56)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x2003;&#x2265;</named-content>40</td><td align="left" valign="top">34 (44)</td></tr><tr><td align="left" valign="top" colspan="2">Sex</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">43 (56)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">34 (44)</td></tr><tr><td align="left" valign="top" colspan="2">Race</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>People of color (Black, Asian, and other)</td><td align="left" valign="top">27 (35)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>White</td><td align="left" valign="top">50 (65)</td></tr><tr><td align="left" valign="top" colspan="2">Treatment location</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Inpatient</td><td align="left" valign="top">33 (43)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Outpatient</td><td align="left" valign="top">44 (57)</td></tr></tbody></table></table-wrap></sec><sec id="s3-2"><title>Interrater and Model Agreement</title><p>We computed weighted &#x03BA; values to assess agreement between each pair of human raters and between each model and each human rater (the full matrix is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>). GPT-4o was evaluated under both zero-shot and few-shot prompting strategies, and Llama-4 was evaluated under a zero-shot prompting strategy.</p><p>All 3 human rater pairs had similar interrater reliability (&#x03BA;=0.77-0.78). Among model-human comparisons, the GPT-4o zero-shot model showed the highest agreement with individual raters (&#x03BA;=0.83, 95% CI 0.75-0.88), followed closely by the GPT-4o few-shot model (&#x03BA;=0.78, 95% CI 0.69-0.85), suggesting that additional prompting examples did not meaningfully improve agreement. The remaining GPT-4o model-human agreement pairings approached the level of human-human agreement (&#x03BA;=0.70, 95% CI 0.58-0.80 vs &#x03BA;=0.77, 95% CI 0.66-0.84), while Llama-4 zero-shot agreement with individual raters was more variable and lower than that of the GPT-4o models: &#x03BA;=0.68 (95% CI 0.57&#x2010;0.77) with R1, &#x03BA;=0.50 (95% CI 0.39&#x2010;0.61) with R2, and &#x03BA;=0.52 (95% CI 0.41&#x2010;0.63) with R3. When comparing models directly, the GPT-4o zero-shot and few-shot models showed high mutual agreement (&#x03BA;=0.89, 95% CI 0.82&#x2010;0.93), whereas Llama-4 demonstrated substantial but slightly lower agreement with GPT-4o outputs (zero-shot comparison: &#x03BA;=0.79, 95% CI 0.71&#x2010;0.85; few-shot comparison: &#x03BA;=0.79, 95% CI 0.70&#x2010;0.85).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Pairwise weighted Cohen &#x03BA; values between human raters and large language models (LLMs). Pairwise weighted Cohen &#x03BA; values showing agreement among 3 human raters (R1, R2, and R3) and between each rater and each LLM. GPT-4o was evaluated under both zero-shot and few-shot prompting conditions, and Llama-4 was evaluated under a zero-shot prompting condition. Each cell represents agreement between a pair of raters or models. &#x03BA; values between 0.40 and 0.60 indicate moderate agreement, values between 0.60 and 0.80 indicate substantial agreement, and values &#x003E;0.80 indicate excellent agreement. Cells display weighted Cohen &#x03BA; values; values in parentheses indicate 95% CIs estimated via nonparametric bootstrap resampling (5000 iterations).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e86906_fig01.png"/></fig></sec><sec id="s3-3"><title>LLM Agreement With Human Consensus Ratings</title><p>We next compared model-predicted CGI scores to a reference standard based on human consensus ratings (<xref ref-type="table" rid="table2">Table 2</xref>).</p><p>Both the GPT-4o zero-shot (&#x03BA;=0.85, 95% CI 0.78-0.90) and few-shot (&#x03BA;=0.84, 95% CI 0.77-0.89) models showed robust agreement with the average of the human ratings, exceeding the agreement observed between either model or any individual human rater. Llama-4 zero-shot agreement with the average human rating was &#x03BA;=0.70 (95% CI 0.55&#x2010;0.80), indicating moderate to substantial agreement but lower concordance than GPT-4o. Because model-human agreement is inherently constrained by the reliability of the human reference standard, we further examined performance across 2 subsets of notes: those for which all 3 raters agreed (n=22, 28.6%) and those on which at least 2 of the 3 raters agreed (n=68, 88.3%). For notes with unanimous human agreement, GPT-4o performance remained high (zero-shot: &#x03BA;=0.88, 95% CI 0.77&#x2010;0.94; few-shot: &#x03BA;=0.90, 95% CI 0.80&#x2010;0.96), while Llama-4 demonstrated substantial agreement (&#x03BA;=0.72, 95% CI 0.53&#x2010;0.83). For notes with majority agreement, GPT-4o agreement decreased slightly (zero-shot: &#x03BA;=0.79, 95% CI 0.70-0.85; few-shot: &#x03BA;=0.77, 95% CI 0.66&#x2010;0.84), approaching the level of human-human reliability (&#x03BA;=0.77-0.78). Llama-4 agreement in this subset was also lower, with &#x03BA;=0.60 (95% CI 0.48-0.72), consistent with moderate agreement. Given that averaging across multiple ratings improves measurement reliability [<xref ref-type="bibr" rid="ref52">52</xref>-<xref ref-type="bibr" rid="ref54">54</xref>] and that the GPT-4o zero-shot model generally performed better than both the GPT-4o few-shot and Llama-4 models, all subsequent analyses were conducted using the GPT-4o zero-shot model and the average of the human ratings.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Weighted Cohen &#x03BA; between model predictions and consensus human ratings.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Compared to clinician ratings</td><td align="left" valign="bottom">GPT-4o: zero-shot, &#x03BA; (95% CI)</td><td align="left" valign="bottom">GPT-4o: few-shot, &#x03BA; (95% CI)</td><td align="left" valign="bottom">Llama-4: zero-shot, &#x03BA; (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Average rating (all notes; N=77)</td><td align="left" valign="top">0.85 (0.78-0.90)</td><td align="left" valign="top">0.84 (0.77-0.89)</td><td align="char" char="." valign="top">0.70 (0.55&#x2010;0.80)</td></tr><tr><td align="left" valign="top">Unanimous rating (notes where all 3 raters agreed; n=22)</td><td align="left" valign="top">0.88 (0.77-0.94)</td><td align="left" valign="top">0.90 (0.80-0.96)</td><td align="char" char="." valign="top">0.72 (0.53-0.83)</td></tr><tr><td align="left" valign="top">Majority rating (notes where &#x2265;2 raters agreed; n=68)</td><td align="left" valign="top">0.79 (0.70-0.85)</td><td align="left" valign="top">0.77 (0.66-0.84)</td><td align="char" char="." valign="top">0.60 (0.48-0.72)</td></tr></tbody></table></table-wrap><p><xref ref-type="fig" rid="figure2">Figure 2</xref> presents the confusion matrix illustrating agreement between the GPT-4o zero-shot model and the average of the human ratings for individual notes. The zero-shot model predictions were largely within &#x00B1;1 CGI point of the average human rating, with no instances of &#x2265;2-point underestimation and a single instance of &#x2265;2-point overestimation. Among discordant cases (n=31, 40.3%), discrepancies more frequently reflected overestimation by the LLM (n=22, 28.6%) than underestimation (n=9, 11.7%). To evaluate whether directional errors were associated with patient- or note-level characteristics, we conducted exploratory univariable logistic regression analyses separately examining 2 outcomes: LLM overestimation vs all other notes and LLM underestimation vs all other notes. Covariates tested included age, sex, race, treatment location, note length dichotomized at the median, and the percentage of copy-forwarded text also dichotomized at the median. No variables were significantly associated with LLM overestimation or LLM underestimation, although the sample sizes for these comparisons were small, and caution is warranted when interpreting these findings.</p><p>To further evaluate the directional errors, we conducted an additional qualitative review of discrepant cases. This review suggested that underestimations were typically driven by notes emphasizing an unremarkable mental status examination, with only brief phrases indicating symptoms in the remainder of the note text (eg, &#x201C;some instability&#x201D; and &#x201C;low at times&#x201D;). Conversely, overestimations appeared to be driven by disproportionate emphasis on 1 or 2 specific mental status findings, which may have overshadowed a relatively less severe overall clinical assessment based on interval history or patient report. Additionally, mentions of anxiety, despite the prompt being focused on depressive symptoms, and the inpatient treatment setting were common factors associated with model overestimation.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Confusion matrix comparing average human ratings with GPT-4o zero-shot model predictions. Confusion matrix showing agreement between the GPT-4o zero-shot model and the average of the human ratings for individual clinical notes. Each cell represents the number of notes assigned a given Clinical Global Impression (CGI) scale score by the model (columns) vs the average human rating (rows). Diagonal cells represent perfect agreement; off-diagonal cells indicate discrepancies, which were predominantly within 1 scale point, with a single instance of &#x2265;2-point overestimation observed.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e86906_fig02.png"/></fig></sec><sec id="s3-4"><title>Subgroup Analysis</title><p>We next carried out exploratory analyses to assess whether the extent of agreement between the GPT-4o zero-shot model and the average human ratings differed according to the age, sex, or race of the patient; the clinical location; note length; or the percentage of copy-forwarded text in the note. <xref ref-type="table" rid="table3">Table 3</xref> reports the weighted &#x03BA; values for each subgroup along with the empirical significance of the difference between strata. Agreement remained high across most subgroups, although model-human agreement differed significantly by note length. Notes below the median length had lower agreement with the average human ratings than notes at or above the median length (&#x03BA;=0.72 vs 0.92; <italic>P</italic>=.003). However, agreement did not significantly differ by the percentage of copy-forwarded text (&#x03BA;=0.86 vs 0.81; <italic>P</italic>=.43).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Weighted Cohen &#x03BA; across subgroups and permutation tests for differences.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Demographic characteristics</td><td align="left" valign="bottom">Weighted Cohen &#x03BA;</td><td align="left" valign="bottom">Permutation test, <italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Age (years)</td><td align="char" char="." valign="top">.39</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x003C;40</td><td align="left" valign="top">0.82</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x2003;&#x2265;</named-content>40</td><td align="left" valign="top">0.87</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2">Sex</td><td align="char" char="." valign="top">.38</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">0.88</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">0.82</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2">Race</td><td align="char" char="." valign="top">.74</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Non-White</td><td align="left" valign="top">0.83</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>White</td><td align="left" valign="top">0.86</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2">Treatment location</td><td align="char" char="." valign="top">.58</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Inpatient</td><td align="left" valign="top">0.83</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Outpatient</td><td align="left" valign="top">0.86</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2">Word count</td><td align="char" char="." valign="top">.003</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Below median (&#x003C;1007 words)</td><td align="left" valign="top">0.72</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>At or above median (&#x2265;1007 words)</td><td align="left" valign="top">0.92</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2">Copy-forwarded text</td><td align="char" char="." valign="top">.43</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Below median (&#x003C;81.3%)</td><td align="left" valign="top">0.86</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>At or above median (&#x2265;81.3%)</td><td align="left" valign="top">0.81</td><td align="left" valign="top"/></tr></tbody></table></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>In this study, we sought to assess whether LLMs could produce reliable CGI scores, a clinician-rated measure of current patient symptom burden and impairment, from psychiatric clinical notes and how these scores compare with psychiatrist ratings. Using a prompt engineering strategy without fine-tuning, GPT-4o produced CGI scores that closely matched clinician ratings. Agreement between the model-generated and clinician-assigned CGI scores was comparable to CGI interrater reliability values previously reported among human raters (ICC or &#x03BA;&#x2248;0.7-0.8) [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref29">29</xref>] and was especially high when compared with the average of all human raters (&#x03BA;=0.84-0.85). Comparing the model against the average may inflate the estimate of &#x03BA;, but it provides a useful indication of the upper bound of the model&#x2019;s performance. Notably, the GPT-4o model performed markedly better than a leading open-source model, and few-shot prompting did not demonstrate improved agreement compared with zero-shot prompting. Although a formal statistical test comparing the zero-shot and few-shot models was not performed, the point estimates were nearly identical, with substantial overlap in their 95% CIs. These findings suggest that a clear, task-specific prompt anchored to the CGI rubric may be sufficient to achieve strong performance with GPT-4o.</p><p>Two key design choices were made to optimize model performance. First, we explicitly aligned the prompt with an established guide for CGI scoring. This design choice was intentional to standardize the rating task between the model and human raters; however, performance may not generalize to settings in which the rubric is omitted or modified. Further work is needed to evaluate whether similar performance is maintained across alternative rubrics or psychiatric conditions other than depression. We also included language instructing the model to prioritize recent information in the clinical notes, to ensure that the generated scores reflected the patient&#x2019;s current status rather than past clinical history. The latter may be especially important given the frequency with which clinical notes are copy-forwarded, with studies estimating that approximately half of note content is duplicated from the author&#x2019;s previous note [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]. The phenomenon of copy-forward within the clinical notes could potentially bias the model, without appropriate remediation, toward prioritizing prior information and lead it to fail to detect more recent clinical changes. Second, for few-shot prompting, we selected examples from notes with complete human rater agreement to maximize gold-standard fidelity. The lack of incremental benefit from few-shot over zero-shot prompting, however, may reflect a ceiling effect of our rubric-based prompt. Because human interrater reliability in our sample was &#x03BA;=0.77&#x2010;0.78, model-human agreement would be expected to plateau near this level, even though agreement with a consensus gold-standard reference could be higher.</p><p>To assess model performance across potential subgroups, we further carried out stratified analyses by age, sex, race, treatment location, note length, and copy-forwarded text burden. Agreement remained high across most subgroups, but it did differ significantly by note length, with notes at or above the median length having higher agreement with average human ratings than those below the median length. This suggests that the amount of available clinical context may influence model performance, with shorter notes providing fewer details from which the model can infer depression severity. Given this finding, future work with LLM-derived CGI scores may need to be paired with minimum note length or information-sufficiency thresholds and may require human review when notes are brief or clinically sparse. This interpretation is consistent with our qualitative review of discrepant cases, in which some errors appeared to arise when brief symptom descriptions or isolated mental status findings were weighed disproportionately relative to the overall clinical impression. In contrast, agreement did not significantly differ by copy-forward burden, despite the high prevalence of copy-forwarded text in the sample. Although copy-forwarded text could theoretically bias the model toward prior clinical information rather than the patient&#x2019;s current status, this finding suggests that prompting the model to prioritize recent information may have helped mitigate this concern. In exploratory error analyses, we also examined whether patient- or note-level characteristics were associated with the direction of model disagreement and found that no characteristics were significantly associated with LLM overestimation or underestimation, although these analyses were limited by the modest number of discordant cases. Together, these findings suggest generally consistent model performance across the measured subgroups while underscoring the need for larger studies to evaluate potential sources of systematic error.</p><p>Our study has several important limitations. First, we demonstrated only the reliability of an LLM compared with psychiatrists in assigning CGI scores based on information available in the text of clinical notes. This approach does not consider other information available in the EHR, such as diagnostic subtypes, clinical comorbidities, and treatment history, which may contribute to the global clinical impression of a patient. Moreover, it is unclear to what extent CGI scores based on clinical notes would compare with CGI scores assigned by psychiatrists based on their direct clinical evaluation of the patients (ie, concurrent validity) or with other measures of depression-related constructs such as the PHQ-9 or MADRS (ie, convergent validity). Unfortunately, we did not have sufficient data to examine the correlations between model-based CGI scores and other concurrent measures of depression. Importantly, LLM-generated CGI scores from clinical notes are not intended to replace clinical judgment or standardized symptom scales such as the PHQ-9 or MADRS, which serve complementary roles in clinical assessment and MBC. The relationship between CGI ratings and these scales has been well established in the prior literature, reflecting related but distinct constructs, and the present study does not aim to re-evaluate the validity of the CGI or to compare these instruments. Future work will be needed to evaluate concurrent validity by comparing text-derived CGI scores with clinician-assigned CGI ratings obtained at the time of care; however, such analyses were beyond the scope of the present study. Second, the study was limited to encounters with a primary diagnosis of MDD within a single hospital system and to notes authored by psychiatrists. Consequently, the findings may not be generalizable to other diagnoses, settings, or types of clinical providers, where varying documentation styles are practiced. Third, the 3 psychiatrist raters were study coauthors and were not blinded to the overall study objective, so potential expectancy effects cannot be fully excluded despite independent annotation procedures. This is a common limitation of annotation-based studies and should be addressed in future work using external raters independent of the study team. Finally, we observed the best performance with a closed-source LLM, which raises concerns about reproducibility, version stability over time, and cost feasibility. Performance may vary across model versions due to model updates or drift, and access to proprietary models may be constrained in some research or clinical settings. Due to these limitations, more work is needed to establish the validity of using LLMs to assign CGI scores based on clinical notes, especially in comparison with the concurrent assessments of psychiatrists based on direct clinical evaluation, and to determine the generalizability of the approach to other psychiatric diagnoses and clinical settings. Our study is an important first step, and the findings demonstrate that this future work is warranted.</p><p>These findings extend prior work showing that NLP methods, especially LLMs, can be used to support patient phenotyping and outcome measurement such as depression response or remission [<xref ref-type="bibr" rid="ref33">33</xref>], suicide risk [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref47">47</xref>], and dimensional symptom ratings [<xref ref-type="bibr" rid="ref48">48</xref>]. Our study demonstrates that LLMs can generate CGI scores, a clinician-rated global measure that is validated and widely used in clinical research but rarely captured in EHRs, based on clinical notes with agreement comparable to psychiatrist ratings. Because CGI ratings have shown good interrater reliability and strong correlations with other standardized measures [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref57">57</xref>], the ability to generate CGI scores from narrative clinical notes offers, if further validated, the potential of a pragmatic approach to generate outcome measures suitable for real-world studies using EHR data, as well as open up the possibility of deriving such measures in real time from ambient AI technology to facilitate MBC.</p></sec></body><back><ack><p>The authors declare that generative AI was not used in the creation of any portion of this manuscript. All authors attest to the integrity and accuracy of this work.</p></ack><notes><sec><title>Funding</title><p>Research reported in this publication was partially funded through a Patient-Centered Outcomes Research Institute (PCORI) Award (ME-2020C3-21145; principal investigator: EAS). This work was also supported by the Stanley and Elizabeth Star Precision Medicine Center of Excellence in Mood Disorders, which was established with a generous gift from the Star family.</p></sec><sec><title>Data Availability</title><p>The analytic code and summary data used in this study are available from the corresponding author upon reasonable request. The original electronic health record data, including clinical notes, cannot be shared publicly due to institutional and legal restrictions under the HIPAA (Health Insurance Portability and Accountability Act).</p></sec></notes><fn-group><fn fn-type="conflict"><p>EAS reports consulting income from Eli Lilly and Company and from serving on behalf of plaintiffs in litigation regarding genital talc exposure and ovarian cancer, for work unrelated to this study. JPG receives support as the Associate Director of the Train New Trainers Primary Care Fellowship (University of California, Irvine), an educational program that equips primary care clinicians to recognize and manage common mental health conditions. All other authors declare no other conflicts of interest.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BDI</term><def><p>Beck Depression Inventory</p></def></def-item><def-item><term id="abb2">CGI</term><def><p>Clinical Global Impression</p></def></def-item><def-item><term id="abb3">CGI-I</term><def><p>Clinical Global Impression&#x2013;Improvement</p></def></def-item><def-item><term id="abb4">CGI-S</term><def><p>Clinical Global Impression&#x2013;Severity</p></def></def-item><def-item><term id="abb5">DASS-21</term><def><p>Depression Anxiety Stress Scales&#x2013;21</p></def></def-item><def-item><term id="abb6">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb7">H&#x0026;P</term><def><p>history and physical</p></def></def-item><def-item><term id="abb8">HAM-D</term><def><p>Hamilton Depression Rating Scale</p></def></def-item><def-item><term id="abb9">HIPAA</term><def><p>Health Insurance Portability and Accountability Act</p></def></def-item><def-item><term id="abb10">HoNOS</term><def><p>Health of the Nation Outcome Scales</p></def></def-item><def-item><term id="abb11">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb12"><italic>ICD-10</italic></term><def><p><italic>International Statistical Classification of Diseases and Related Health Problems, tenth revision</italic></p></def></def-item><def-item><term id="abb13">iCGI</term><def><p>improved Clinical Global Impression</p></def></def-item><def-item><term id="abb14">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb15">JHBMC</term><def><p>Johns Hopkins Bayview Medical Center</p></def></def-item><def-item><term id="abb16">JHH</term><def><p>Johns Hopkins Hospital</p></def></def-item><def-item><term id="abb17">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb18">MADRS</term><def><p>Montgomery-&#x00C5;sberg Depression Rating Scale</p></def></def-item><def-item><term id="abb19">MBC</term><def><p>measurement-based care</p></def></def-item><def-item><term id="abb20">MDD</term><def><p>major depressive disorder</p></def></def-item><def-item><term id="abb21">MHQ-14</term><def><p>Mental Health Questionnaire&#x2013;14</p></def></def-item><def-item><term id="abb22">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb23">PHI</term><def><p>protected health information</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Global, regional, and national temporal trend in burden of major depressive disorder from 1990 to 2019: an analysis of the Global Burden of Disease study</article-title><source>Psychiatry Res</source><year>2024</year><month>07</month><volume>337</volume><fpage>115958</fpage><pub-id pub-id-type="doi">10.1016/j.psychres.2024.115958</pub-id><pub-id pub-id-type="medline">38772160</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Qiao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Kan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name></person-group><article-title>Global, regional, and national burden of depression, 1990-2021: a decomposition and age-period-cohort analysis with projection to 2040</article-title><source>J Affect Disord</source><year>2025</year><month>12</month><day>15</day><volume>391</volume><fpage>120018</fpage><pub-id pub-id-type="doi">10.1016/j.jad.2025.120018</pub-id><pub-id pub-id-type="medline">40782921</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nemeroff</surname><given-names>CB</given-names> </name></person-group><article-title>The state of our understanding of the pathophysiology and optimal treatment of depression: glass half full or half empty?</article-title><source>Am J Psychiatry</source><year>2020</year><month>08</month><day>1</day><volume>177</volume><issue>8</issue><fpage>671</fpage><lpage>685</lpage><pub-id pub-id-type="doi">10.1176/appi.ajp.2020.20060845</pub-id><pub-id pub-id-type="medline">32741287</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lundberg</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cars</surname><given-names>T</given-names> </name><name name-style="western"><surname>L&#x00F6;&#x00F6;v</surname><given-names>S&#x00C5;</given-names> </name><etal/></person-group><article-title>Association of treatment-resistant depression with patient outcomes and health care resource utilization in a population-wide study</article-title><source>JAMA Psychiatry</source><year>2023</year><month>02</month><day>1</day><volume>80</volume><issue>2</issue><fpage>167</fpage><lpage>175</lpage><pub-id pub-id-type="doi">10.1001/jamapsychiatry.2022.3860</pub-id><pub-id pub-id-type="medline">36515938</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maj</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stein</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Parker</surname><given-names>G</given-names> </name><etal/></person-group><article-title>The clinical characterization of the adult patient with depression aimed at personalization of management</article-title><source>World Psychiatry</source><year>2020</year><month>10</month><volume>19</volume><issue>3</issue><fpage>269</fpage><lpage>293</lpage><pub-id pub-id-type="doi">10.1002/wps.20771</pub-id><pub-id pub-id-type="medline">32931110</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Boyd</surname><given-names>M</given-names> </name><name name-style="western"><surname>Puspitasari</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Implementing measurement-based care in behavioral health: a review</article-title><source>JAMA Psychiatry</source><year>2019</year><month>03</month><day>1</day><volume>76</volume><issue>3</issue><fpage>324</fpage><lpage>335</lpage><pub-id pub-id-type="doi">10.1001/jamapsychiatry.2018.3329</pub-id><pub-id pub-id-type="medline">30566197</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Glazer</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rootes-Murdy</surname><given-names>K</given-names> </name><name name-style="western"><surname>Van Wert</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mondimore</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zandi</surname><given-names>P</given-names> </name></person-group><article-title>The utility of PHQ-9 and CGI-S in measurement-based care for predicting suicidal ideation and behaviors</article-title><source>J Affect Disord</source><year>2020</year><month>04</month><day>1</day><volume>266</volume><fpage>766</fpage><lpage>771</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2018.05.054</pub-id><pub-id pub-id-type="medline">29954612</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cerimele</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Goldberg</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Gabrielson</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Fortney</surname><given-names>JC</given-names> </name></person-group><article-title>Systematic review of symptom assessment measures for use in measurement-based care of bipolar disorders</article-title><source>Psychiatr Serv</source><year>2019</year><month>05</month><day>1</day><volume>70</volume><issue>5</issue><fpage>396</fpage><lpage>408</lpage><pub-id pub-id-type="doi">10.1176/appi.ps.201800383</pub-id><pub-id pub-id-type="medline">30717645</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cheung</surname><given-names>BS</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Michalak</surname><given-names>EE</given-names> </name><etal/></person-group><article-title>Barriers and facilitators to technology-enhanced measurement based care for depression among Canadian clinicians and patients: results of an online survey</article-title><source>J Affect Disord</source><year>2023</year><month>01</month><day>1</day><volume>320</volume><fpage>1</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2022.09.055</pub-id><pub-id pub-id-type="medline">36162664</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zandi</surname><given-names>PP</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>PD</given-names> </name><etal/></person-group><article-title>Development of the National Network of Depression Centers Mood Outcomes Program: a multisite platform for measurement-based care</article-title><source>Psychiatr Serv</source><year>2020</year><month>05</month><day>1</day><volume>71</volume><issue>5</issue><fpage>456</fpage><lpage>464</lpage><pub-id pub-id-type="doi">10.1176/appi.ps.201900481</pub-id><pub-id pub-id-type="medline">31960777</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Busner</surname><given-names>J</given-names> </name><name name-style="western"><surname>Targum</surname><given-names>SD</given-names> </name></person-group><article-title>The Clinical Global Impressions scale: applying a research tool in clinical practice</article-title><source>Psychiatry (Edgmont)</source><year>2007</year><month>07</month><volume>4</volume><issue>7</issue><fpage>28</fpage><lpage>37</lpage><pub-id pub-id-type="medline">20526405</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Berk</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>F</given-names> </name><name name-style="western"><surname>Dodd</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The validity of the CGI severity and improvement scales as measures of clinical effectiveness suitable for routine clinical use</article-title><source>J Eval Clin Pract</source><year>2008</year><month>12</month><volume>14</volume><issue>6</issue><fpage>979</fpage><lpage>983</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2753.2007.00921.x</pub-id><pub-id pub-id-type="medline">18462279</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Forkmann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Scherer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Boecker</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pawelzik</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jostes</surname><given-names>R</given-names> </name><name name-style="western"><surname>Gauggel</surname><given-names>S</given-names> </name></person-group><article-title>The Clinical Global Impression scale and the influence of patient or staff perspective on outcome</article-title><source>BMC Psychiatry</source><year>2011</year><month>05</month><day>14</day><volume>11</volume><fpage>83</fpage><pub-id pub-id-type="doi">10.1186/1471-244X-11-83</pub-id><pub-id pub-id-type="medline">21569566</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bschor</surname><given-names>T</given-names> </name><name name-style="western"><surname>Nagel</surname><given-names>L</given-names> </name><name name-style="western"><surname>Unger</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schwarzer</surname><given-names>G</given-names> </name><name name-style="western"><surname>Baethge</surname><given-names>C</given-names> </name></person-group><article-title>Differential outcomes of placebo treatment across 9 psychiatric disorders: a systematic review and meta-analysis</article-title><source>JAMA Psychiatry</source><year>2024</year><month>08</month><day>1</day><volume>81</volume><issue>8</issue><fpage>757</fpage><lpage>768</lpage><pub-id pub-id-type="doi">10.1001/jamapsychiatry.2024.0994</pub-id><pub-id pub-id-type="medline">38809560</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mohebbi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dodd</surname><given-names>S</given-names> </name><name name-style="western"><surname>Dean</surname><given-names>OM</given-names> </name><name name-style="western"><surname>Berk</surname><given-names>M</given-names> </name></person-group><article-title>Patient centric measures for a patient centric era: agreement and convergent between ratings on the Patient Global Impression of Improvement (PGI-I) scale and the Clinical Global Impressions - Improvement (CGI-I) scale in bipolar and major depressive disorder</article-title><source>Eur Psychiatry</source><year>2018</year><month>09</month><volume>53</volume><fpage>17</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.1016/j.eurpsy.2018.05.006</pub-id><pub-id pub-id-type="medline">29859377</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Magalh&#x00E3;es</surname><given-names>PV</given-names> </name><name name-style="western"><surname>Manzolli</surname><given-names>P</given-names> </name><name name-style="western"><surname>Walz</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Kapczinski</surname><given-names>F</given-names> </name></person-group><article-title>A bidimensional solution for outcomes in bipolar disorder</article-title><source>J Nerv Ment Dis</source><year>2012</year><month>02</month><volume>200</volume><issue>2</issue><fpage>180</fpage><lpage>182</lpage><pub-id pub-id-type="doi">10.1097/NMD.0b013e3182439885</pub-id><pub-id pub-id-type="medline">22297318</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Assis da Silva</surname><given-names>R</given-names> </name><name name-style="western"><surname>Mograbi</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Silveira</surname><given-names>LA</given-names> </name><etal/></person-group><article-title>The reliability of self-assessment of affective state in different phases of bipolar disorder</article-title><source>J Nerv Ment Dis</source><year>2014</year><month>05</month><volume>202</volume><issue>5</issue><fpage>386</fpage><lpage>390</lpage><pub-id pub-id-type="doi">10.1097/NMD.0000000000000136</pub-id><pub-id pub-id-type="medline">24727726</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Denicoff</surname><given-names>KD</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>SO</given-names> </name><name name-style="western"><surname>Sollinger</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Smith-Jackson</surname><given-names>EE</given-names> </name><name name-style="western"><surname>Leverich</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Post</surname><given-names>RM</given-names> </name></person-group><article-title>Utility of the daily prospective National Institute of Mental Health Life-Chart Method (NIMH-LCM-p) ratings in clinical trials of bipolar disorder</article-title><source>Depress Anxiety</source><year>2002</year><volume>15</volume><issue>1</issue><fpage>1</fpage><lpage>9</lpage><pub-id pub-id-type="doi">10.1002/da.1078</pub-id><pub-id pub-id-type="medline">11816046</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bourredjem</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pelissolo</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rotge</surname><given-names>JY</given-names> </name><etal/></person-group><article-title>A video Clinical Global Impression (CGI) in obsessive compulsive disorder</article-title><source>Psychiatry Res</source><year>2011</year><month>03</month><day>30</day><volume>186</volume><issue>1</issue><fpage>117</fpage><lpage>122</lpage><pub-id pub-id-type="doi">10.1016/j.psychres.2010.06.021</pub-id><pub-id pub-id-type="medline">20621362</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zaider</surname><given-names>TI</given-names> </name><name name-style="western"><surname>Heimberg</surname><given-names>RG</given-names> </name><name name-style="western"><surname>Fresco</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Schneier</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Liebowitz</surname><given-names>MR</given-names> </name></person-group><article-title>Evaluation of the Clinical Global Impression scale among individuals with social anxiety disorder</article-title><source>Psychol Med</source><year>2003</year><month>05</month><volume>33</volume><issue>4</issue><fpage>611</fpage><lpage>622</lpage><pub-id pub-id-type="doi">10.1017/s0033291703007414</pub-id><pub-id pub-id-type="medline">12785463</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khau</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tabbane</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bloom</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Pragmatic implementation of the Clinical Global Impression Scale of Severity as a tool for measurement-based care in a first-episode psychosis program</article-title><source>Schizophr Res</source><year>2022</year><month>05</month><volume>243</volume><fpage>147</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.1016/j.schres.2022.03.007</pub-id><pub-id pub-id-type="medline">35339824</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goldman</surname><given-names>M</given-names> </name><name name-style="western"><surname>DeQuardo</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Tandon</surname><given-names>R</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Jibson</surname><given-names>M</given-names> </name></person-group><article-title>Symptom correlates of global measures of severity in schizophrenia</article-title><source>Compr Psychiatry</source><year>1999</year><volume>40</volume><issue>6</issue><fpage>458</fpage><lpage>461</lpage><pub-id pub-id-type="doi">10.1016/s0010-440x(99)90090-1</pub-id><pub-id pub-id-type="medline">10579378</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leucht</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kane</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Etschel</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kissling</surname><given-names>W</given-names> </name><name name-style="western"><surname>Hamann</surname><given-names>J</given-names> </name><name name-style="western"><surname>Engel</surname><given-names>RR</given-names> </name></person-group><article-title>Linking the PANSS, BPRS, and CGI: clinical implications</article-title><source>Neuropsychopharmacology</source><year>2006</year><month>10</month><volume>31</volume><issue>10</issue><fpage>2318</fpage><lpage>2325</lpage><pub-id pub-id-type="doi">10.1038/sj.npp.1301147</pub-id><pub-id pub-id-type="medline">16823384</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leucht</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fennema</surname><given-names>H</given-names> </name><name name-style="western"><surname>Engel</surname><given-names>RR</given-names> </name><name name-style="western"><surname>Kaspers-Janssen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lepping</surname><given-names>P</given-names> </name><name name-style="western"><surname>Szegedi</surname><given-names>A</given-names> </name></person-group><article-title>What does the MADRS mean? Equipercentile linking with the CGI using a company database of mirtazapine studies</article-title><source>J Affect Disord</source><year>2017</year><month>03</month><day>1</day><volume>210</volume><fpage>287</fpage><lpage>293</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2016.12.041</pub-id><pub-id pub-id-type="medline">28068617</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Shankles</surname><given-names>EB</given-names> </name><name name-style="western"><surname>Polissar</surname><given-names>NL</given-names> </name></person-group><article-title>Relative sensitivity of the Montgomery-Asberg Depression Rating Scale, the Hamilton Depression rating scale and the Clinical Global Impressions rating scale in antidepressant clinical trials</article-title><source>Int Clin Psychopharmacol</source><year>2002</year><month>11</month><volume>17</volume><issue>6</issue><fpage>281</fpage><lpage>285</lpage><pub-id pub-id-type="doi">10.1097/00004850-200211000-00003</pub-id><pub-id pub-id-type="medline">12409681</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Spielmans</surname><given-names>GI</given-names> </name><name name-style="western"><surname>McFall</surname><given-names>JP</given-names> </name></person-group><article-title>A comparative meta-analysis of Clinical Global Impressions change in antidepressant trials</article-title><source>J Nerv Ment Dis</source><year>2006</year><month>11</month><volume>194</volume><issue>11</issue><fpage>845</fpage><lpage>852</lpage><pub-id pub-id-type="doi">10.1097/01.nmd.0000244554.91259.27</pub-id><pub-id pub-id-type="medline">17102709</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kadouri</surname><given-names>A</given-names> </name><name name-style="western"><surname>Corruble</surname><given-names>E</given-names> </name><name name-style="western"><surname>Falissard</surname><given-names>B</given-names> </name></person-group><article-title>The improved Clinical Global Impression scale (iCGI): development and validation in depression</article-title><source>BMC Psychiatry</source><year>2007</year><month>02</month><day>6</day><volume>7</volume><fpage>7</fpage><pub-id pub-id-type="doi">10.1186/1471-244X-7-7</pub-id><pub-id pub-id-type="medline">17284321</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haro</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Kamath</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Ochoa</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The Clinical Global Impression-Schizophrenia scale: a simple instrument to measure the diversity of symptoms present in schizophrenia</article-title><source>Acta Psychiatr Scand Suppl</source><year>2003</year><issue>416</issue><fpage>16</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.1034/j.1600-0447.107.s416.5.x</pub-id><pub-id pub-id-type="medline">12755850</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leon</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Shear</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Klerman</surname><given-names>GL</given-names> </name><name name-style="western"><surname>Portera</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rosenbaum</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Goldenberg</surname><given-names>I</given-names> </name></person-group><article-title>A comparison of symptom determinants of patient and clinician global ratings in patients with panic disorder and depression</article-title><source>J Clin Psychopharmacol</source><year>1993</year><month>10</month><volume>13</volume><issue>5</issue><fpage>327</fpage><lpage>331</lpage><pub-id pub-id-type="medline">8227491</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Targum</surname><given-names>SD</given-names> </name><name name-style="western"><surname>Busner</surname><given-names>J</given-names> </name><name name-style="western"><surname>Young</surname><given-names>AH</given-names> </name></person-group><article-title>Targeted scoring criteria reduce variance in global impressions</article-title><source>Hum Psychopharmacol</source><year>2008</year><month>10</month><volume>23</volume><issue>7</issue><fpage>629</fpage><lpage>633</lpage><pub-id pub-id-type="doi">10.1002/hup.966</pub-id><pub-id pub-id-type="medline">18666094</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beneke</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rasmus</surname><given-names>W</given-names> </name></person-group><article-title>&#x201C;Clinical Global Impressions&#x201D; (ECDEU): some critical comments</article-title><source>Pharmacopsychiatry</source><year>1992</year><month>07</month><volume>25</volume><issue>4</issue><fpage>171</fpage><lpage>176</lpage><pub-id pub-id-type="doi">10.1055/s-2007-1014401</pub-id><pub-id pub-id-type="medline">1528956</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Busner</surname><given-names>J</given-names> </name><name name-style="western"><surname>Targum</surname><given-names>SD</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>DS</given-names> </name></person-group><article-title>The Clinical Global Impressions scale: errors in understanding and use</article-title><source>Compr Psychiatry</source><year>2009</year><volume>50</volume><issue>3</issue><fpage>257</fpage><lpage>262</lpage><pub-id pub-id-type="doi">10.1016/j.comppsych.2008.08.005</pub-id><pub-id pub-id-type="medline">19374971</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name><name name-style="western"><surname>Iosifescu</surname><given-names>DV</given-names> </name><name name-style="western"><surname>Castro</surname><given-names>VM</given-names> </name><etal/></person-group><article-title>Using electronic medical records to enable large-scale studies in psychiatry: treatment resistant depression as a model</article-title><source>Psychol Med</source><year>2012</year><month>01</month><volume>42</volume><issue>1</issue><fpage>41</fpage><lpage>50</lpage><pub-id pub-id-type="doi">10.1017/S0033291711000997</pub-id><pub-id pub-id-type="medline">21682950</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Irving</surname><given-names>J</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>R</given-names> </name><name name-style="western"><surname>Oliver</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Using natural language processing on electronic health records to enhance detection and prediction of psychosis risk</article-title><source>Schizophr Bull</source><year>2021</year><month>03</month><day>16</day><volume>47</volume><issue>2</issue><fpage>405</fpage><lpage>414</lpage><pub-id pub-id-type="doi">10.1093/schbul/sbaa126</pub-id><pub-id pub-id-type="medline">33025017</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Castro</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Minnier</surname><given-names>J</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>SN</given-names> </name><etal/></person-group><article-title>Validation of electronic health record phenotyping of bipolar disorder cases and controls</article-title><source>Am J Psychiatry</source><year>2015</year><month>04</month><volume>172</volume><issue>4</issue><fpage>363</fpage><lpage>372</lpage><pub-id pub-id-type="doi">10.1176/appi.ajp.2014.14030423</pub-id><pub-id pub-id-type="medline">25827034</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCoy</surname><given-names>TH</given-names>  <suffix>Jr</suffix></name><name name-style="western"><surname>Yu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hart</surname><given-names>KL</given-names> </name><etal/></person-group><article-title>High throughput phenotyping for dimensional psychopathology in electronic health records</article-title><source>Biol Psychiatry</source><year>2018</year><month>06</month><day>15</day><volume>83</volume><issue>12</issue><fpage>997</fpage><lpage>1004</lpage><pub-id pub-id-type="doi">10.1016/j.biopsych.2018.01.011</pub-id><pub-id pub-id-type="medline">29496195</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCoy</surname><given-names>TH Jr</given-names> </name><name name-style="western"><surname>Castro</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Roberson</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Snapper</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name></person-group><article-title>Improving prediction of suicide and accidental death after discharge from general hospitals with natural language processing</article-title><source>JAMA Psychiatry</source><year>2016</year><month>10</month><day>1</day><volume>73</volume><issue>10</issue><fpage>1064</fpage><lpage>1071</lpage><pub-id pub-id-type="doi">10.1001/jamapsychiatry.2016.2172</pub-id><pub-id pub-id-type="medline">27626235</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Morgan</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Diederen</surname><given-names>K</given-names> </name><name name-style="western"><surname>V&#x00E9;rtes</surname><given-names>PE</given-names> </name><etal/></person-group><article-title>Natural language processing markers in first episode psychosis and people at clinical high-risk</article-title><source>Transl Psychiatry</source><year>2021</year><month>12</month><day>13</day><volume>11</volume><issue>1</issue><fpage>630</fpage><pub-id pub-id-type="doi">10.1038/s41398-021-01722-y</pub-id><pub-id pub-id-type="medline">34903724</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Soffer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Charney</surname><given-names>AW</given-names> </name><name name-style="western"><surname>Landi</surname><given-names>I</given-names> </name><name name-style="western"><surname>Nadkarni</surname><given-names>GN</given-names> </name><name name-style="western"><surname>Klang</surname><given-names>E</given-names> </name></person-group><article-title>Applications of large language models in psychiatry: a systematic review</article-title><source>Front Psychiatry</source><year>2024</year><volume>15</volume><fpage>1422807</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2024.1422807</pub-id><pub-id pub-id-type="medline">38979501</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Volkmer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Meyer-Lindenberg</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schwarz</surname><given-names>E</given-names> </name></person-group><article-title>Large language models in psychiatry: opportunities and challenges</article-title><source>Psychiatry Res</source><year>2024</year><month>09</month><volume>339</volume><fpage>116026</fpage><pub-id pub-id-type="doi">10.1016/j.psychres.2024.116026</pub-id><pub-id pub-id-type="medline">38909412</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mosteiro</surname><given-names>P</given-names> </name><name name-style="western"><surname>Rijcken</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zervanou</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kaymak</surname><given-names>U</given-names> </name><name name-style="western"><surname>Scheepers</surname><given-names>F</given-names> </name><name name-style="western"><surname>Spruit</surname><given-names>M</given-names> </name></person-group><article-title>Machine learning for violence risk assessment using Dutch clinical notes</article-title><source>J Artif Intell Med Sci</source><year>2021</year><volume>2</volume><fpage>44</fpage><lpage>54</lpage><pub-id pub-id-type="doi">10.2991/jaims.d.210225.001</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>LY</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>XC</given-names> </name><name name-style="western"><surname>Nejatian</surname><given-names>NP</given-names> </name><etal/></person-group><article-title>Health system-scale language models are all-purpose prediction engines</article-title><source>Nature</source><year>2023</year><month>07</month><volume>619</volume><issue>7969</issue><fpage>357</fpage><lpage>362</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06160-y</pub-id><pub-id pub-id-type="medline">37286606</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gargari</surname><given-names>OK</given-names> </name><name name-style="western"><surname>Fatehi</surname><given-names>F</given-names> </name><name name-style="western"><surname>Mohammadi</surname><given-names>I</given-names> </name><name name-style="western"><surname>Firouzabadi</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Shafiee</surname><given-names>A</given-names> </name><name name-style="western"><surname>Habibi</surname><given-names>G</given-names> </name></person-group><article-title>Diagnostic accuracy of large language models in psychiatry</article-title><source>Asian J Psychiatr</source><year>2024</year><month>10</month><volume>100</volume><fpage>104168</fpage><pub-id pub-id-type="doi">10.1016/j.ajp.2024.104168</pub-id><pub-id pub-id-type="medline">39111087</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Kao</surname><given-names>YC</given-names> </name><name name-style="western"><surname>Tsai</surname><given-names>SJ</given-names> </name><etal/></person-group><article-title>Comparing the performance of ChatGPT GPT-4, Bard, and Llama-2 in the Taiwan Psychiatric Licensing Examination and in differential diagnosis with multi-center psychiatrists</article-title><source>Psychiatry Clin Neurosci</source><year>2024</year><month>06</month><volume>78</volume><issue>6</issue><fpage>347</fpage><lpage>352</lpage><pub-id pub-id-type="doi">10.1111/pcn.13656</pub-id><pub-id pub-id-type="medline">38404249</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCoy</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Castro</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name></person-group><article-title>Estimating depression severity in narrative clinical notes using large language models</article-title><source>J Affect Disord</source><year>2025</year><month>07</month><day>15</day><volume>381</volume><fpage>270</fpage><lpage>274</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2025.04.014</pub-id><pub-id pub-id-type="medline">40187432</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wiest</surname><given-names>IC</given-names> </name><name name-style="western"><surname>Verhees</surname><given-names>FG</given-names> </name><name name-style="western"><surname>Ferber</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Detection of suicidality from medical text using privacy-preserving large language models</article-title><source>Br J Psychiatry</source><year>2024</year><month>12</month><volume>225</volume><issue>6</issue><fpage>532</fpage><lpage>537</lpage><pub-id pub-id-type="doi">10.1192/bjp.2024.134</pub-id><pub-id pub-id-type="medline">39497458</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCoy</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name></person-group><article-title>Reasoning language models for more transparent prediction of suicide risk</article-title><source>BMJ Ment Health</source><year>2025</year><month>05</month><day>11</day><volume>28</volume><issue>1</issue><fpage>e301654</fpage><pub-id pub-id-type="doi">10.1136/bmjment-2025-301654</pub-id><pub-id pub-id-type="medline">40350181</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCoy</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name></person-group><article-title>Dimensional measures of psychopathology in children and adolescents using large language models</article-title><source>Biol Psychiatry</source><year>2024</year><month>12</month><day>15</day><volume>96</volume><issue>12</issue><fpage>940</fpage><lpage>947</lpage><pub-id pub-id-type="doi">10.1016/j.biopsych.2024.05.008</pub-id><pub-id pub-id-type="medline">38866172</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Rotondi</surname><given-names>MA</given-names> </name></person-group><article-title>KappaSize: sample size estimation functions for studies of interobserver agreement</article-title><source>The Comprehensive R Archive Network</source><year>2018</year><access-date>2026-07-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/web/packages/kappaSize/index.html">https://cran.r-project.org/web/packages/kappaSize/index.html</ext-link></comment></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>The measurement of observer agreement for categorical data</article-title><source>Biometrics</source><year>1977</year><month>03</month><volume>33</volume><issue>1</issue><fpage>159</fpage><lpage>174</lpage><pub-id pub-id-type="medline">843571</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Gamer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lemon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fellows</surname><given-names>I</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>P</given-names> </name></person-group><article-title>Irr: various coefficients of interrater reliability and agreement</article-title><source>The Comprehensive R Archive Network</source><year>2026</year><access-date>2026-07-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/web/packages/irr/index.html">https://cran.r-project.org/web/packages/irr/index.html</ext-link></comment></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Vet</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Mokkink</surname><given-names>LB</given-names> </name><name name-style="western"><surname>Mosmuller</surname><given-names>DG</given-names> </name><name name-style="western"><surname>Terwee</surname><given-names>CB</given-names> </name></person-group><article-title>Spearman-Brown prophecy formula and Cronbach&#x2019;s alpha: different faces of reliability and opportunities for new applications</article-title><source>J Clin Epidemiol</source><year>2017</year><month>05</month><volume>85</volume><fpage>45</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2017.01.013</pub-id><pub-id pub-id-type="medline">28342902</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leon</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Marzuk</surname><given-names>PM</given-names> </name><name name-style="western"><surname>Portera</surname><given-names>L</given-names> </name></person-group><article-title>More reliable outcome measures can reduce sample size requirements</article-title><source>Arch Gen Psychiatry</source><year>1995</year><month>10</month><volume>52</volume><issue>10</issue><fpage>867</fpage><lpage>871</lpage><pub-id pub-id-type="doi">10.1001/archpsyc.1995.03950220077014</pub-id><pub-id pub-id-type="medline">7575107</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Perkins</surname><given-names>DO</given-names> </name><name name-style="western"><surname>Wyatt</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Bartko</surname><given-names>JJ</given-names> </name></person-group><article-title>Penny-wise and pound-foolish: the impact of measurement error on sample size requirements in clinical trials</article-title><source>Biol Psychiatry</source><year>2000</year><month>04</month><day>15</day><volume>47</volume><issue>8</issue><fpage>762</fpage><lpage>766</lpage><pub-id pub-id-type="doi">10.1016/s0006-3223(00)00837-4</pub-id><pub-id pub-id-type="medline">10773186</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rule</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bedrick</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Hribar</surname><given-names>MR</given-names> </name></person-group><article-title>Length and redundancy of outpatient progress notes across a decade at an academic medical center</article-title><source>JAMA Netw Open</source><year>2021</year><month>07</month><day>1</day><volume>4</volume><issue>7</issue><fpage>e2115334</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2021.15334</pub-id><pub-id pub-id-type="medline">34279650</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steinkamp</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kantrowitz</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Airan-Javia</surname><given-names>S</given-names> </name></person-group><article-title>Prevalence and sources of duplicate information in the electronic medical record</article-title><source>JAMA Netw Open</source><year>2022</year><month>09</month><day>1</day><volume>5</volume><issue>9</issue><fpage>e2233348</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2022.33348</pub-id><pub-id pub-id-type="medline">36156143</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Voort</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Yuruk</surname><given-names>D</given-names> </name><etal/></person-group><article-title>A characterization of the Clinical Global Impressions scale thresholds in the treatment of adolescent depression across multiple rating scales</article-title><source>J Child Adolesc Psychopharmacol</source><year>2022</year><month>06</month><volume>32</volume><issue>5</issue><fpage>278</fpage><lpage>287</lpage><pub-id pub-id-type="doi">10.1089/cap.2021.0111</pub-id><pub-id pub-id-type="medline">35704877</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Clinical Global Impression zero-shot prompt for GPT-4o.</p><media xlink:href="formative_v10i1e86906_app1.docx" xlink:title="DOCX File, 15 KB"/></supplementary-material></app-group></back></article>