<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e94454</article-id><article-id pub-id-type="doi">10.2196/94454</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Two-Stage Extraction of Clinical Course Information From Psychiatric Discharge Summaries Using Fine-Tuned Large Language Models: Cross-Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Chien-Hung</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tseng</surname><given-names>Po-Chang</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Dai</surname><given-names>Hong-Jie</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Su</surname><given-names>Chu-Hsien</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Shi-Heng</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chien</surname><given-names>Yi-Ling</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Huang</surname><given-names>Wei-Lieh</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Wu</surname><given-names>Chi-Shin</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Hsin-Hsi</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff7">7</xref></contrib></contrib-group><aff id="aff1"><institution>Graduate Institute of Networking and Multimedia, National Taiwan University</institution><addr-line>Taipei</addr-line><country>Taiwan</country></aff><aff id="aff2"><institution>Institute of Psychiatry, Psychology &#x0026; Neuroscience, King's College London</institution><addr-line>London</addr-line><country>United Kingdom</country></aff><aff id="aff3"><institution>Department of Electrical Engineering, College of Electrical Engineering and Computer Science, National Kaohsiung University of Science and Technology</institution><addr-line>Kaohsiung</addr-line><country>Taiwan</country></aff><aff id="aff4"><institution>National Center for Geriatrics and Welfare Research, National Health Research Institutes</institution><addr-line>35, Keyan Road</addr-line><addr-line>Zhunan</addr-line><addr-line>Miaoli County</addr-line><country>Taiwan</country></aff><aff id="aff5"><institution>Department of Psychiatry, National Taiwan University Hospital</institution><addr-line>Taipei</addr-line><country>Taiwan</country></aff><aff id="aff6"><institution>Department of Psychiatry, National Taiwan University Hospital, Yunlin Branch</institution><addr-line>Douliu</addr-line><country>Taiwan</country></aff><aff id="aff7"><institution>Department of Computer Science and Information Engineering, National Taiwan University</institution><addr-line>Taipei</addr-line><country>Taiwan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Sharma</surname><given-names>Arun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Xiang</surname><given-names>Jiawei</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Chi-Shin Wu, MD, PhD, National Center for Geriatrics and Welfare Research, National Health Research Institutes, 35, Keyan Road, Zhunan, Miaoli County, 350401, Taiwan, 886 037-206-166 ext 51001; <email>wuchishin@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>6</day><month>8</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e94454</elocation-id><history><date date-type="received"><day>02</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>10</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>13</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Chien-Hung Chen, Po-Chang Tseng, Hong-Jie Dai, Chu-Hsien Su, Shi-Heng Wang, Yi-Ling Chien, Wei-Lieh Huang, Chi-Shin Wu, Hsin-Hsi Chen. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 6.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e94454"/><abstract><sec><title>Background</title><p>Clinical course information, including disease onset, episode recurrence, and hospitalization history, is essential for psychiatric care and research. However, these data are embedded in unstructured clinical narratives with substantial linguistic variability, making manual extraction labor-intensive and rule-based extraction difficult to scale. Fine-tuned large language models (LLMs) may flexibly extract such information from privacy-sensitive psychiatric records.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the performance of LLMs for automatically extracting temporal and clinical course information from psychiatric discharge summaries.</p></sec><sec sec-type="methods"><title>Methods</title><p>We analyzed 500 psychiatric discharge summaries from the Integrated Medical Database of National Taiwan University Hospital. A psychiatrist and a natural language processing researcher manually annotated clinical events and temporal information. Four open-source LLMs (LLaMA, MentaLLaMA, OpenBioLLM, and Mistral) were fine-tuned using low-rank adaptation and evaluated using 10-fold cross-validation. A 2-stage framework was developed in which sentence-level extraction of clinical events and temporal information was followed by chart-level prediction of 4 clinical course features: first-episode onset time, episode count, number of psychiatric hospitalizations, and most recent hospitalization. Performance was assessed using precision, recall, <italic>F</italic><sub>1</sub>-score, accuracy, mean absolute error (MAE), and bootstrap significance testing.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 12,947 sentences were extracted from 500 discharge summaries, yielding 7177 clinical event annotations and 4842 temporal annotations. At the sentence level, Mistral achieved the highest <italic>F</italic><sub>1</sub>-scores for clinical event extraction, including symptom/episode detection (0.925, 95% CI 0.920&#x2010;0.930), hospitalization detection (0.944, 95% CI 0.934&#x2010;0.952), and remission/response detection (0.867, 95% CI 0.851&#x2010;0.883). Mistral also achieved the highest <italic>F</italic><sub>1</sub>-scores for most temporal information categories, including age expressions (0.983, 95% CI 0.973&#x2010;0.993), relative time expressions (0.953, 95% CI 0.944&#x2010;0.963), duration expressions (0.976, 95% CI 0.962&#x2010;0.987), and vague temporal expressions (0.888, 95% CI 0.871&#x2010;0.905). At the chart level, the proposed 2-stage framework showed the clearest benefit for first-episode onset prediction. Using the 2-stage framework, Mistral achieved the highest point estimate for onset accuracy (0.772), although its performance did not significantly differ from that of LLaMA using the same framework (0.744; bootstrap <italic>P</italic>=.26). For onset prediction, the 2-stage framework significantly outperformed the direct and joint extraction approaches across all evaluated models (all bootstrap <italic>P</italic>&#x2264;.002). For other chart-level features, the best-performing approach varied: using the 2-stage framework, Mistral achieved the highest episode count accuracy (0.624; MAE=0.428); using the direct approach, LLaMA achieved the highest hospitalization count accuracy (0.692; MAE=0.379); and using the 2-stage framework, OpenBioLLM achieved the highest most recent hospitalization accuracy (0.868; <italic>F</italic><sub>1</sub>-score=0.617).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Fine-tuned, locally deployable, open-source LLMs can extract temporal and longitudinal disease course information from psychiatric discharge summaries. The 2-stage framework was most beneficial for first-episode onset prediction and performed competitively across other chart-level features, supporting the use of LLMs to transform heterogeneous psychiatric narratives into structured data for research and decision support.</p></sec></abstract><kwd-group><kwd>natural language processing</kwd><kwd>mental disorders</kwd><kwd>medical record linkage</kwd><kwd>medical records</kwd><kwd>time factors</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Understanding the clinical course of psychiatric disorders is essential for effective clinical decision-making and treatment planning [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Clinical course information includes key temporal and event-based details, such as the onset date, the number of past episodes or relapses, and the frequency of psychiatric hospitalizations. Extracting this information from free-text notes is challenging due to the various ways it is documented (eg, &#x201C;first psychotic break at 19,&#x201D; &#x201C;multiple depressive episodes since 2015,&#x201D; and <italic>&#x201C;</italic>no prior psychiatric hospitalizations<italic>&#x201D;</italic>). Manual retrieval from medical charts is both time-consuming and labor-intensive. Natural language processing (NLP) offers a solution by automating the extraction of these critical details, thereby improving efficiency and reducing human error [<xref ref-type="bibr" rid="ref4">4</xref>]. Early NLP approaches combined rule-based algorithms with machine learning. For instance, 1 study developed a hybrid NLP system that extracted time expressions and classified relevant text to generate a ranked timeline of probable psychosis onset dates from mental health records [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>Clinical course information in psychiatric discharge summaries is often expressed in unstructured free text with substantial linguistic variability. For example, the same clinical concept may be documented as &#x201C;first psychotic break at age 19,&#x201D; &#x201C;onset in 2009,&#x201D; or &#x201C;symptoms dating back to his early twenties.&#x201D; Rule-based NLP systems require manually specified patterns for different expression types and may not generalize well to heterogeneous descriptions of psychiatric symptoms, episodes, hospitalizations, and temporal relationships. In addition, although the discharge summaries in this study were predominantly written in English, some records contained Chinese expressions, transliterated terms, or code-mixed content, which further complicated purely rule-based extraction. Large language models (LLMs) are well-suited to this task because they can interpret variable free-text expressions via instruction-following and can be adapted to domain-specific annotation schemas through fine-tuning, as demonstrated by the extraction performance observed in this study. Recent studies demonstrate that LLMs can effectively parse clinical text and answer such questions with reasonable accuracy [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. One study used zero-shot learning with Flan-T5 to extract specifiers, such as severity and remission, for substance use disorders, outperforming traditional rule-based methods in recall by capturing diverse linguistic variations [<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>This study aimed to evaluate locally deployable, open-source LLMs for extracting clinical course information from psychiatric discharge summaries. We developed a 2-stage framework that first extracts sentence-level clinical events and temporal information and then predicts 4 chart-level features: first episode onset time, episode count, number of psychiatric hospitalizations, and most recent hospitalization time. We compared this framework with direct and joint extraction approaches to assess whether explicit sentence-level extraction improves chart-level prediction.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Data Collection</title><p>This study used the Integrated Medical Database of National Taiwan University Hospital (NTUH-iMD), a protected electronic medical record system. These data are not publicly available due to patient privacy and ethical restrictions, as stated in the Data Availability Statement. The NTUH-iMD includes electronic medical records such as admission and discharge notes as well as <italic>ICD</italic> (<italic>International Classification of Diseases</italic>) diagnostic codes. This study focused on 4836 discharge summaries from 3087 patients admitted to the psychiatric unit with a principal psychiatric diagnosis (<italic>ICD-9-CM</italic> [<italic>International Classification of Diseases, Ninth Revision, Clinical Modification</italic>]: 290&#x2010;319 or <italic>ICD-10-CM</italic> [<italic>International Classification of Diseases, Tenth Revision, Clinical Modification</italic>]: F00-F99) [<xref ref-type="bibr" rid="ref10">10</xref>]. Discharge summaries in NTUH-iMD are predominantly written in English, consistent with routine clinical documentation practices at NTUH. Although some records contain limited Chinese expressions, transliterated terms, or code-mixed content, the corpus is best characterized as predominantly English clinical text with occasional multilingual elements rather than Chinese-language records. All personal identifying information, including names, addresses, and birthdays, were deidentified, and only sections related to the history of the present illness were included.</p><p>Due to the time-consuming nature of manual annotation, 500 discharge notes were selected. As previously described [<xref ref-type="bibr" rid="ref10">10</xref>], the dataset included 189 (37.8%) notes on schizophrenia, 109 (21.8%) on bipolar disorder, and 117 (23.4%) on unipolar depressive disorder. Four (0.8%) patients had both schizophrenia and unipolar depressive disorder, and 8 (1.6%) had both schizophrenia and bipolar disorder. Additionally, 77 notes covered other diagnoses&#x2014;dementia, substance use disorder, and anxiety disorders&#x2014;that were not classified under the main 3 categories [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>While basic personal information had already been deidentified, we applied an additional deidentification tool [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>] to replace sensitive details, such as residence, school, workplace, and other elements, that could potentially disclose aspects of the patient&#x2019;s private life. This tool uses advanced NLP to automatically detect and replace sensitive information [<xref ref-type="bibr" rid="ref11">11</xref>]. Although effective, its random replacements can affect data completeness and readability. Therefore, after applying the tool, manual review and modification were conducted to ensure both deidentification and the integrity of the data for subsequent analysis.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study was approved by the Institutional Review Board of National Taiwan University Hospital (NTUH-201610072RINA). The requirement for informed consent was waived because the study used deidentified retrospective electronic medical records. All data were processed and analyzed in accordance with institutional privacy and data protection regulations.</p></sec><sec id="s2-3"><title>Feature Determination</title><p>At the sentence level, key clinical course features were identified, including symptoms/episodes indicating disease onset, relapse, recurrence, remission/response, and hospitalization details. Symptoms/episodes encompassed the onset or worsening of psychiatric symptoms and dangerous behaviors such as suicide or violence. Hospitalization included both inpatient treatment and rehabilitation programs. Time information was extracted in formats such as year-month-date, age, vague terms (eg, &#x201C;childhood&#x201D; or &#x201C;this year&#x201D;), and duration formats (eg, &#x201C;from A to B,&#x201D; &#x201C;between A and B,&#x201D; or &#x201C;persisted for A&#x201D;). These definitions are outlined in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Key clinical course features with examples of clinical events and time formats.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Note segment</td><td align="left" valign="bottom">Event</td><td align="left" valign="bottom">Time</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Event</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>She began to suffer from low mood, insomnia, and suicidal ideation in 2002.</td><td align="left" valign="top">Symptom/episode</td><td align="left" valign="top">Time_YMD: 2002</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Under Solian to 600 mg, her symptoms subsided smoothly, with much less olfactory, gustatory, and somatic hallucination, less disturbing behavior, less social withdrawal, and regular daily activity.</td><td align="left" valign="top">Remission/response</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Therefore, he was admitted to the acute ward in TCPC<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> in February 2010 for 46 days.</td><td align="left" valign="top">Hospitalization</td><td align="left" valign="top">Time_YMD: 2010&#x2010;02; persistence: 46 days</td></tr><tr><td align="left" valign="top" colspan="3">Time</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>At the age of 18 years, the patient developed a major depressive episode with mood-congruent psychotic features, precipitated by partner relational problems (the boyfriend had extramarital relations).</td><td align="left" valign="top">Symptom/episode</td><td align="left" valign="top">Age: 18</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>During the past 2 to 3 weeks, her mood went down again, and she had suicidal ideation (by drug overdose).</td><td align="left" valign="top">Symptom/episode</td><td align="left" valign="top">Ago: 14&#x2010;21 days</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>On March 18, 2007, due to poor drug adherence, her psychotic symptoms aggravated.</td><td align="left" valign="top">Symptom/episode</td><td align="left" valign="top">Time_YMD: 2007-03-18</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Besides, dysphoric, labile, easily agitated mood, poor appetite, initial insomnia, and disturbed sleep-wake cycle, loss of energy and interest were also noted in recent months.</td><td align="left" valign="top">Symptom/episode</td><td align="left" valign="top">Vague: recent months</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Her depression had been left untreated for 3 to 4 years with gradual but complete improvement.</td><td align="left" valign="top">Remission/response</td><td align="left" valign="top">Persistence: 3&#x2010;4 years</td></tr><tr><td align="left" valign="top">&#x2003;So she was admitted again from April 11, 2012, to May 27, 2012.</td><td align="left" valign="top">Hospitalization</td><td align="left" valign="top">Duration: April 11, 2012, to May 27, 2012</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Not available.</p></fn><fn id="table1fn2"><p><sup>b</sup>TCPC: Taipei City Psychiatric Center.</p></fn></table-wrap-foot></table-wrap><p>At the chart level, we aimed to extract 4 key pieces of information:</p><list list-type="order"><list-item><p>Onset time: the first appearance of psychiatric symptoms.</p></list-item><list-item><p>Episode count: the number of times the patient has experienced psychiatric symptoms, with each episode defined as the period from onset to full or partial remission.</p></list-item><list-item><p>Number of psychiatric hospitalizations: the total number of psychiatric hospitalizations includes admissions to acute and chronic wards and rehabilitation centers.</p></list-item><list-item><p>Most recent hospitalization time: the time of the most recent psychiatric hospitalization. If no such information is available, it will be recorded as &#x201C;none.&#x201D;</p></list-item></list><p>Some charts did not provide precise episode counts or hospitalization details (eg, &#x201C;The patient had several episodes in the past 10 years&#x201D;). These were excluded from the corresponding analysis.</p></sec><sec id="s2-4"><title>Annotation Process</title><p>All medical chart annotations were conducted by a board-certified clinical psychiatrist with 20 years of experience (CSW) and a researcher with NLP expertise (CHS). Annotators were instructed to label each sentence based on content, identifying the development or exacerbation of psychiatric symptoms, behaviors related to suicide or violence as &#x201C;symptom/episode,&#x201D; treatment response or remission as &#x201C;response/remission,&#x201D; admissions to acute psychiatric or rehabilitation wards as &#x201C;hospitalization,&#x201D; or to mark the sentence as containing none of the above events. Timing information was also recorded. For example, in the sentence &#x201C;He had insomnia and poor appetite for one month,&#x201D; symptoms were annotated along with their duration. Additionally, chart data&#x2014;including the first onset time, episode count, number of psychiatric hospitalizations, and the date of the most recent hospitalization&#x2014;were annotated. To assess annotation reliability, a psychiatrist and an NLP researcher independently annotated all training documents. Interrater agreement was evaluated using Cohen &#x03BA; before consensus adjudication. Disagreements were subsequently resolved through discussion to generate the final reference standard used for model development and evaluation.</p><p>Overall interrater agreement was substantial (Cohen &#x03BA;=0.78). Interrater agreement varied across annotation categories (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Agreement was highest for explicit temporal expressions, including calendar dates (&#x03BA;=0.90), age-based expressions (&#x03BA;=0.85), and hospitalization events (&#x03BA;=0.87). In contrast, vague temporal expressions showed substantially lower agreement (&#x03BA;=0.47), reflecting the inherent ambiguity of phrases such as &#x201C;during childhood,&#x201D; &#x201C;recent years,&#x201D; or &#x201C;a long time ago.&#x201D;</p></sec><sec id="s2-5"><title>Model Development</title><p>Due to the sensitive nature of psychiatric notes and privacy concerns, we opted for locally deployable models instead of cloud-based LLMs. We used the following models: LLaMA (Llama-3.1-8B-Instruct) [<xref ref-type="bibr" rid="ref13">13</xref>], MentaLLaMA [<xref ref-type="bibr" rid="ref14">14</xref>], OpenBioLLM [<xref ref-type="bibr" rid="ref15">15</xref>], and Mistral (Ministral-8B-Instruct-2410) [<xref ref-type="bibr" rid="ref16">16</xref>]. We chose these models because they are popular and can be fine-tuned on consumer-level hardware. In addition to the LLMs, we included Bio_ClinicalBERT as a conventional transformer-based baseline for sentence-level information extraction. Bio_ClinicalBERT is a clinical-domain BERT model widely used for clinical named entity recognition tasks [<xref ref-type="bibr" rid="ref17">17</xref>]. Because Bio_ClinicalBERT is a sequence-labeling model rather than a generative model, it was evaluated only for sentence-level extraction tasks and was not included in chart-level prediction analyses.</p><p>Our approach employs a 2-stage extraction method to systematically derive clinical information from psychiatric medical records as illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>. Stage 1&#x2014;sentence-level extraction: For each sentence in a discharge note, the sentence-level extractor produces: (1) an event label (&#x201C;symptom/episode,&#x201D; &#x201C;hospitalization,&#x201D; &#x201C;remission/response,&#x201D; or &#x201C;none&#x201D;) and (2) associated temporal information (if present), in plain-text format. Stage 2&#x2014;chart-level aggregation and prediction: all sentence-level predictions are assembled into a structured list maintaining original document order, with the format: [SENTENCE]: &#x003C;text &#x003E;| [EVENT]: &#x003C;label &#x003E;| [TIME]: &#x003C;temporal info or &#x201C;none&#x201D;&#x003E;. This structured list is provided as input to the Stage 2 chart&#x2014;level extractor, which synthesizes this information to produce the 4 clinical course features (onset time, episode count, hospitalization count, and last hospitalization time).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>End-to-end pipeline diagram.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e94454_fig01.png"/></fig><p>For sentence-level extraction, we develop a specialized system prompt that clearly communicates the task objective: accurately classifying medical and temporal event types from sentences in clinical notes, as shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>. The design of this prompt focuses on guiding the model to reliably identify and categorize relevant medical events, including episodes, symptoms, hospitalizations, and temporal indicators. By emphasizing context and temporal cues, the prompt facilitates precise extraction, ensuring consistency across varying clinical scenarios.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>System prompt for the sentence-level extractor.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e94454_fig02.png"/></fig><p>In the second stage, the chart-level extractor synthesizes aggregated sentence-level predictions to derive detailed psychiatric summaries from patient records. The chart-level prompt was crafted to systematically guide the extraction of key clinical features, including the first occurrence of symptoms, the number of episodes, hospitalization count, and the timing of the most recent hospitalization. This design aims to standardize the summarization process, clarify the criteria for data extraction, and minimize ambiguity, thereby enabling robust, consistent results suitable for clinical analysis. The detailed system prompt is shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>.</p><p>Both extractors use low-rank adaptation (LoRA) [<xref ref-type="bibr" rid="ref18">18</xref>] for fine-tuning. Annotated data were explicitly structured for fine-tuning, converting labeled annotations into corresponding prompt-response pairs aligned with each designed prompt. This structured data approach facilitates supervised learning and improves model performance.</p><p>The fine-tuning configurations were optimized to ensure training efficiency and accuracy. The batch size was set to 1, with gradient accumulation across 16 steps. We set the learning rate to 0.0001 and trained the model for 5 epochs, using a warm-up ratio of 0.1. The maximum sequence length was set to 8192 tokens. For LoRA, we set the rank to 8, the &#x03B1; to 16, and applied a dropout rate of 0.05. The LoRA hyperparameters were selected via a grid search over rank &#x2208; {4, 8, 16}, alpha &#x2208; {8, 16, 32}, and dropout &#x2208; {0.01, 0.05, 0.1}, with validation performance averaged across the first 3 folds.</p><p>All experiments were conducted on a single NVIDIA A6000 48 GB GPU. Fine-tuning each model for 1 fold required approximately 2 to 3 hours (sentence-level extractor: ~1.5 h; chart-level extractor: ~1 h). Inference (prediction) took approximately 20 minutes per fold per model. Models can run on consumer hardware with &#x2265;24 GB of VRAM using 4-bit quantization, though fine-tuning would require proportionally more time.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>System prompt for the chart-level extractor.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e94454_fig03.png"/></fig></sec><sec id="s2-6"><title>Evaluation Metrics and Statistical Comparison</title><p>Sentence-level extraction performance was evaluated separately for clinical events and temporal information. Precision, recall, and <italic>F</italic><sub>1</sub>-score were calculated for each annotation category. Performance was estimated using 10-fold cross-validation, and the mean performance across folds was reported. Clinical event extraction was evaluated for symptom/episode, hospitalization, and remission/response events. Temporal extraction was evaluated for calendar dates, age expressions, relative time expressions, duration, persistence, and vague temporal expressions.</p><p>Chart-level performance was evaluated for four clinical course features: first-episode onset time, number of episodes, number of hospitalizations, and most recent hospitalization. For temporal outcomes (first-episode onset time and most recent hospitalization), performance was assessed using exact-match accuracy and macroaveraged <italic>F</italic><sub>1</sub>-score. Predicted temporal expressions were normalized prior to evaluation, and partial matches (eg, &#x201C;2010&#x201D; vs &#x201C;2010&#x2010;03&#x201D;) were considered incorrect. For count-based outcomes (number of episodes and number of hospitalizations), both exact-match accuracy and mean absolute error (MAE) were reported. Lower MAE values indicate better performance.</p><p>To assess whether performance differences between methods were statistically meaningful, pairwise comparisons were conducted using bootstrap resampling. For each comparison, 1000 bootstrap samples were generated by resampling discharge summaries with replacement from the evaluation set, and 2-sided bootstrap <italic>P</italic> values were calculated from the resulting distribution of performance differences. Comparisons focused on the proposed 2-stage framework vs the direct and Joint approaches.</p></sec><sec id="s2-7"><title>Additional Analyses</title><sec id="s2-7-1"><title>Oracle Analysis</title><p>To quantify the contribution of sentence-level extraction errors to overall chart-level performance, an oracle analysis was performed. In this analysis, gold-standard sentence-level annotations were provided directly to the chart-level extraction stage, replacing model-generated sentence predictions. The resulting performance represents an approximate upper bound of the second-stage chart-level extractor under perfect sentence-level extraction.</p></sec><sec id="s2-7-2"><title>Prompt Sensitivity Analysis</title><p>A prompt sensitivity analysis was conducted to evaluate the effects of prompt design choices. Four prompt variants were examined: (1) the initial structured prompt, (2) a version without few-shot examples, (3) a version incorporating chain-of-thought reasoning instructions, and (4) the final simplified plain-text output format used in the proposed framework. Sentence-level <italic>F</italic><sub>1</sub>-scores were compared across prompt variants.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Data Collection</title><p>In an analysis of 500 discharge notes, a total of 12,947 sentences were extracted, with sentence counts ranging from 4 to 112 per note, averaging 25.9 (SD 17.3) sentences. Of these, 7177 sentences contained clinical events, and 4842 included time-related information.</p></sec><sec id="s3-2"><title>Sentence-Level Information Extraction Results</title><p>Fine-tuning substantially improved sentence-level extraction performance across all evaluated LLMs (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). While performance gains varied across models, LoRA fine-tuning consistently increased <italic>F</italic><sub>1</sub>-scores for symptom/episode, hospitalization, and remission/response extraction. The largest improvements were observed in OpenBioLLM and MentaLLaMA, which demonstrated limited zero-shot performance but achieved competitive results after fine-tuning. These findings underscore the importance of task-specific adaptation when applying LLMs to psychiatric clinical narratives.</p><p><xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref> present the sentence-level extraction performance of all models after fine-tuning. Mistral achieved the highest <italic>F</italic><sub>1</sub>-scores across most clinical event categories, including symptom/episode (0.925), hospitalization (0.944), and remission/response (0.867). It also demonstrated the strongest performance for most temporal information categories, including age (0.983), relative time expressions (0.953), duration (0.976), time references (0.991), and vague temporal expressions (0.888). Bio_ClinicalBERT, a conventional transformer-based sequence-labeling model, achieved competitive performance for several extraction tasks but was generally outperformed by the best-performing fine-tuned LLMs. Nevertheless, LLaMA and OpenBioLLM showed performance comparable to Mistral for duration and time references. Persistent and vague temporal expressions remained the most challenging categories across all models, although Mistral achieved the highest performance in both tasks.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Sentence-level performance for clinical event extraction across models<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="3">Symptom/episode</td><td align="left" valign="bottom" colspan="3">Hospitalization</td><td align="left" valign="bottom" colspan="3">Remission/response</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">LLaMA</td><td align="left" valign="top">0.875</td><td align="left" valign="top">0.885</td><td align="left" valign="top">0.880 (0.874&#x2010;0.886)</td><td align="left" valign="top">0.945</td><td align="left" valign="top">0.896</td><td align="left" valign="top">0.920 (0.908&#x2010;0.931)</td><td align="left" valign="top">0.795</td><td align="left" valign="top">0.786</td><td align="left" valign="top">0.790 (0.771&#x2010;0.808)</td></tr><tr><td align="left" valign="top">MentaLLaMA</td><td align="left" valign="top">0.875</td><td align="left" valign="top">0.875</td><td align="left" valign="top">0.875 (0.869&#x2010;0.882)</td><td align="left" valign="top">0.902</td><td align="left" valign="top">0.909</td><td align="left" valign="top">0.905 (0.894&#x2010;0.916)</td><td align="left" valign="top">0.784</td><td align="left" valign="top">0.763</td><td align="left" valign="top">0.773 (0.753&#x2010;0.794)</td></tr><tr><td align="left" valign="top">OpenBioLLM</td><td align="left" valign="top">0.877</td><td align="left" valign="top">0.867</td><td align="left" valign="top">0.872 (0.865&#x2010;0.878)</td><td align="left" valign="top">0.927</td><td align="left" valign="top">0.913</td><td align="left" valign="top">0.920 (0.909&#x2010;0.930)</td><td align="left" valign="top">0.808</td><td align="left" valign="top">0.753</td><td align="left" valign="top">0.779 (0.759&#x2010;0.799)</td></tr><tr><td align="left" valign="top">Mistral</td><td align="left" valign="top">0.923</td><td align="left" valign="top">0.926</td><td align="left" valign="top">0.925 (0.920&#x2010;0.930)<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">0.982</td><td align="left" valign="top">0.908</td><td align="left" valign="top">0.944 (0.934&#x2010;0.952)<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">0.880</td><td align="left" valign="top">0.854</td><td align="left" valign="top">0.867 (0.851&#x2010;0.883)<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Bio_ClinicalBERT</td><td align="left" valign="top">0.865</td><td align="left" valign="top">0.864</td><td align="left" valign="top">0.864 (0.858&#x2010;0.871)</td><td align="left" valign="top">0.906</td><td align="left" valign="top">0.915</td><td align="left" valign="top">0.910 (0.898&#x2010;0.921)</td><td align="left" valign="top">0.787</td><td align="left" valign="top">0.725</td><td align="left" valign="top">0.755 (0.734&#x2010;0.775)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Higher values indicate better performance. All large language models were fine-tuned using low-rank adaptation and evaluated using the same 10-fold cross-validation framework. </p></fn><fn id="table2fn2"><p><sup>b</sup>Indicates the highest <italic>F</italic><sub>1</sub>-score for each event category.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Sentence-level performance for temporal information extraction across models<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Age <italic>F</italic><sub>1</sub> (95% CI)</td><td align="left" valign="bottom">Ago <italic>F</italic><sub>1</sub> (95% CI)</td><td align="left" valign="bottom">Duration <italic>F</italic><sub>1</sub> (95% CI)</td><td align="left" valign="bottom">Persistence <italic>F</italic><sub>1</sub> (95% CI)</td><td align="left" valign="bottom">Time <italic>F</italic><sub>1</sub> (95% CI)</td><td align="left" valign="bottom">Vague <italic>F</italic><sub>1</sub> (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">LLaMA</td><td align="left" valign="top">0.961 (0.946&#x2010;0.976)</td><td align="left" valign="top">0.926 (0.915&#x2010;0.939)</td><td align="left" valign="top">0.963 (0.947&#x2010;0.977)</td><td align="left" valign="top">0.529 (0.448&#x2010;0.596)</td><td align="left" valign="top">0.988 (0.985&#x2010;0.991)</td><td align="left" valign="top">0.815 (0.791&#x2010;0.837)</td></tr><tr><td align="left" valign="top">MentaLLaMA</td><td align="left" valign="top">0.954 (0.938&#x2010;0.969)</td><td align="left" valign="top">0.925 (0.913&#x2010;0.938)</td><td align="left" valign="top">0.961 (0.945&#x2010;0.975)</td><td align="left" valign="top">0.571 (0.493&#x2010;0.641)</td><td align="left" valign="top">0.988 (0.985&#x2010;0.991)</td><td align="left" valign="top">0.812 (0.788&#x2010;0.835)</td></tr><tr><td align="left" valign="top">OpenBioLLM</td><td align="left" valign="top">0.960 (0.945&#x2010;0.973)</td><td align="left" valign="top">0.927 (0.914&#x2010;0.938)</td><td align="left" valign="top">0.966 (0.951&#x2010;0.981)</td><td align="left" valign="top">0.513 (0.431&#x2010;0.591)</td><td align="left" valign="top">0.989 (0.985&#x2010;0.992)</td><td align="left" valign="top">0.808 (0.783&#x2010;0.829)</td></tr><tr><td align="left" valign="top">Mistral<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">0.983 (0.973&#x2010;0.993)</td><td align="left" valign="top">0.953 (0.944&#x2010;0.963)</td><td align="left" valign="top">0.976 (0.962&#x2010;0.987)</td><td align="left" valign="top">0.740 (0.684&#x2010;0.797)</td><td align="left" valign="top">0.991 (0.988&#x2010;0.993)</td><td align="left" valign="top">0.888 (0.871&#x2010;0.905)</td></tr><tr><td align="left" valign="top">Bio_ClinicalBERT</td><td align="left" valign="top">0.950 (0.932&#x2010;0.965)</td><td align="left" valign="top">0.924 (0.911&#x2010;0.936)</td><td align="left" valign="top">0.969 (0.953&#x2010;0.982)</td><td align="left" valign="top">0.605 (0.532&#x2010;0.665)</td><td align="left" valign="top">0.986 (0.983&#x2010;0.989)</td><td align="left" valign="top">0.797 (0.772&#x2010;0.820)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Values are <italic>F</italic><sub>1</sub>-scores. Higher values indicate better performance. All LLMs were fine-tuned using LoRA and evaluated under the same 10-fold cross-validation framework. </p></fn><fn id="table3fn2"><p><sup>b</sup>The values of Mistral model have the highest <italic>F</italic><sub>1</sub>-score for each temporal category.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Clinical Course Information Extraction Results</title><p><xref ref-type="table" rid="table4">Table 4</xref> summarizes chart-level performance in predicting key clinical course features, whereas <xref ref-type="table" rid="table5">Table 5</xref> presents bootstrap-based pairwise comparisons between the proposed 2-stage framework and alternative approaches. Across all models, the 2-stage framework achieved the highest performance for the first-episode onset prediction. Mistral_Ours achieved the best onset accuracy (0.772), followed by LLaMA_Ours (0.744), OpenBioLLM_Ours (0.738), and MentaLLaMA_Ours (0.734). Detailed precision and recall results for the proposed 2-stage framework (ours) are provided in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>For all 4 models, the 2-stage framework significantly outperformed both the direct and joint approaches for onset prediction (all bootstrap <italic>P</italic>&#x2264;.002). For episode count prediction, Mistral_Ours (accuracy=0.624) and LLaMA_Ours (accuracy=0.618) achieved comparable performance, with no statistically significant difference between them (bootstrap <italic>P</italic>=.85). For hospitalization count prediction, improvements over the direct approach were limited, with no significant differences observed for either LLaMA (<italic>P</italic>=.56) or Mistral (<italic>P</italic>=.99). However, the 2-stage framework consistently outperformed the joint approach for hospitalization count prediction across all evaluated models. For the most recent hospitalization, the 2-stage framework significantly outperformed the direct approach for all models (all bootstrap <italic>P</italic>&#x003C;.05), although differences between the 2-stage and joint approaches were generally smaller.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Chart-level performance for clinical course feature extraction using direct, joint, and 2-stage approaches<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Method</td><td align="left" valign="bottom">Onset accuracy</td><td align="left" valign="bottom">Onset <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Episode count accuracy</td><td align="left" valign="bottom">Episode count MAE<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="bottom">Hospitalization count accuracy</td><td align="left" valign="bottom">Hospitalization count MAE</td><td align="left" valign="bottom">Latest hospitalization accuracy</td><td align="left" valign="bottom">Latest hospitalization <italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">LLaMA_Direct<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">0.63</td><td align="left" valign="top">0.368</td><td align="left" valign="top">0.542</td><td align="left" valign="top">0.532</td><td align="left" valign="top">0.692<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.379<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.754</td><td align="left" valign="top">0.405</td></tr><tr><td align="left" valign="top">LLaMA_Joint<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.426</td><td align="left" valign="top">0.522</td><td align="left" valign="top">0.536</td><td align="left" valign="top">0.584</td><td align="left" valign="top">0.516</td><td align="left" valign="top">0.832</td><td align="left" valign="top">0.538</td></tr><tr><td align="left" valign="top">LLaMA_Ours<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top">0.744</td><td align="left" valign="top">0.547</td><td align="left" valign="top">0.618</td><td align="left" valign="top">0.475</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.452</td><td align="left" valign="top">0.866</td><td align="left" valign="top">0.628<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">MentaLLaMA_Direct</td><td align="left" valign="top">0.558</td><td align="left" valign="top">0.308</td><td align="left" valign="top">0.484</td><td align="left" valign="top">0.653</td><td align="left" valign="top">0.568</td><td align="left" valign="top">0.627</td><td align="left" valign="top">0.742</td><td align="left" valign="top">0.385</td></tr><tr><td align="left" valign="top">MentaLLaMA_Joint</td><td align="left" valign="top">0.484</td><td align="left" valign="top">0.318</td><td align="left" valign="top">0.306</td><td align="left" valign="top">0.831</td><td align="left" valign="top">0.404</td><td align="left" valign="top">0.805</td><td align="left" valign="top">0.734</td><td align="left" valign="top">0.377</td></tr><tr><td align="left" valign="top">MentaLLaMA_Ours</td><td align="left" valign="top">0.734</td><td align="left" valign="top">0.531</td><td align="left" valign="top">0.504</td><td align="left" valign="top">0.598</td><td align="left" valign="top">0.61</td><td align="left" valign="top">0.55</td><td align="left" valign="top">0.858</td><td align="left" valign="top">0.591</td></tr><tr><td align="left" valign="top">OpenBioLLM_Direct</td><td align="left" valign="top">0.596</td><td align="left" valign="top">0.356</td><td align="left" valign="top">0.532</td><td align="left" valign="top">0.578</td><td align="left" valign="top">0.644</td><td align="left" valign="top">0.473</td><td align="left" valign="top">0.788</td><td align="left" valign="top">0.459</td></tr><tr><td align="left" valign="top">OpenBioLLM_Joint</td><td align="left" valign="top">0.644</td><td align="left" valign="top">0.418</td><td align="left" valign="top">0.488</td><td align="left" valign="top">0.588</td><td align="left" valign="top">0.612</td><td align="left" valign="top">0.454</td><td align="left" valign="top">0.838</td><td align="left" valign="top">0.547</td></tr><tr><td align="left" valign="top">OpenBioLLM_Ours</td><td align="left" valign="top">0.738</td><td align="left" valign="top">0.538</td><td align="left" valign="top">0.574</td><td align="left" valign="top">0.483</td><td align="left" valign="top">0.682</td><td align="left" valign="top">0.428</td><td align="left" valign="top">0.868<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.617</td></tr><tr><td align="left" valign="top">Mistral_Direct</td><td align="left" valign="top">0.648</td><td align="left" valign="top">0.388</td><td align="left" valign="top">0.58</td><td align="left" valign="top">0.474</td><td align="left" valign="top">0.674</td><td align="left" valign="top">0.427</td><td align="left" valign="top">0.784</td><td align="left" valign="top">0.436</td></tr><tr><td align="left" valign="top">Mistral_Joint</td><td align="left" valign="top">0.614</td><td align="left" valign="top">0.379</td><td align="left" valign="top">0.522</td><td align="left" valign="top">0.519</td><td align="left" valign="top">0.606</td><td align="left" valign="top">0.388</td><td align="left" valign="top">0.83</td><td align="left" valign="top">0.538</td></tr><tr><td align="left" valign="top">Mistral_Ours</td><td align="left" valign="top">0.772<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.582<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.624<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.428<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.676</td><td align="left" valign="top">0.436</td><td align="left" valign="top">0.858</td><td align="left" valign="top">0.605</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Onset indicates the first-episode onset time and last hospitalization denotes the most recent psychiatric hospitalization. Accuracy and <italic>F</italic><sub>1</sub>-score were used for temporal prediction tasks, whereas both exact-match accuracy and mean absolute error (MAE) were reported for count-based tasks. Higher values indicate better performance for accuracy and <italic>F</italic><sub>1</sub>-score; lower values indicate better performance for MAE. </p></fn><fn id="table4fn2"><p><sup>b</sup>MAE: mean absolute error.</p></fn><fn id="table4fn3"><p><sup>c</sup>Direct: direct chart-level prediction.</p></fn><fn id="table4fn4"><p><sup>d</sup>indicates the best performance for each temporal category.</p></fn><fn id="table4fn5"><p><sup>e</sup>Joint: single-stage joint extraction.</p></fn><fn id="table4fn6"><p><sup>f</sup>Ours: proposed 2-stage framework.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Bootstrap-based pairwise comparisons of chart-level accuracy<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Comparison</td><td align="left" valign="top">Mean (ours)</td><td align="left" valign="top">Mean (baseline)</td><td align="left" valign="top">&#x0394;</td><td align="left" valign="top"><italic>P</italic> (bootstrap)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">Onset time</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaMA_Ours<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup> vs Direct<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td><td align="left" valign="top">0.744</td><td align="left" valign="top">0.630</td><td align="left" valign="top">+0.114</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaMA_Ours vs Joint<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">0.744</td><td align="left" valign="top">0.640</td><td align="left" valign="top">+0.104</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs Direct</td><td align="left" valign="top">0.772</td><td align="left" valign="top">0.648</td><td align="left" valign="top">+0.124</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs Joint</td><td align="left" valign="top">0.772</td><td align="left" valign="top">0.614</td><td align="left" valign="top">+0.158</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>OpenBioLLM_Ours vs Direct</td><td align="left" valign="top">0.738</td><td align="left" valign="top">0.596</td><td align="left" valign="top">+0.142</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>OpenBioLLM_Ours vs Joint</td><td align="left" valign="top">0.738</td><td align="left" valign="top">0.644</td><td align="left" valign="top">+0.094</td><td align="left" valign="top">.002</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MentaLLaMA_Ours vs Direct</td><td align="left" valign="top">0.734</td><td align="left" valign="top">0.558</td><td align="left" valign="top">+0.176</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MentaLLaMA_Ours vs Joint</td><td align="left" valign="top">0.734</td><td align="left" valign="top">0.484</td><td align="left" valign="top">+0.250</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs LLaMA_Ours</td><td align="left" valign="top">0.772</td><td align="left" valign="top">0.744</td><td align="left" valign="top">+0.028</td><td align="left" valign="top">.26</td></tr><tr><td align="left" valign="top" colspan="5">Episode count</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaMA_Ours vs Direct</td><td align="left" valign="top">0.618</td><td align="left" valign="top">0.542</td><td align="left" valign="top">+0.076</td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaMA_Ours vs Joint</td><td align="left" valign="top">0.618</td><td align="left" valign="top">0.522</td><td align="left" valign="top">+0.096</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs Direct</td><td align="left" valign="top">0.624</td><td align="left" valign="top">0.580</td><td align="left" valign="top">+0.044</td><td align="left" valign="top">.27</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs Joint</td><td align="left" valign="top">0.624</td><td align="left" valign="top">0.522</td><td align="left" valign="top">+0.102</td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>OpenBioLLM_Ours vs Joint</td><td align="left" valign="top">0.574</td><td align="left" valign="top">0.488</td><td align="left" valign="top">+0.086</td><td align="left" valign="top">.006</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MentaLLaMA_Ours vs Joint</td><td align="left" valign="top">0.504</td><td align="left" valign="top">0.306</td><td align="left" valign="top">+0.198</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs LLaMA_Ours</td><td align="left" valign="top">0.624</td><td align="left" valign="top">0.618</td><td align="left" valign="top">+0.006</td><td align="left" valign="top">.85</td></tr><tr><td align="left" valign="top" colspan="5">Hospitalization count</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaMA_Ours vs Direct</td><td align="left" valign="top">0.670</td><td align="left" valign="top">0.692</td><td align="left" valign="top">&#x2212;0.022</td><td align="left" valign="top">.56</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaMA_Ours vs Joint</td><td align="left" valign="top">0.670</td><td align="left" valign="top">0.584</td><td align="left" valign="top">+0.086</td><td align="left" valign="top">.02</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs Direct</td><td align="left" valign="top">0.676</td><td align="left" valign="top">0.674</td><td align="left" valign="top">+0.002</td><td align="left" valign="top">.99</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs Joint</td><td align="left" valign="top">0.676</td><td align="left" valign="top">0.606</td><td align="left" valign="top">+0.070</td><td align="left" valign="top">.02</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>OpenBioLLM_Ours vs Joint</td><td align="left" valign="top">0.682</td><td align="left" valign="top">0.612</td><td align="left" valign="top">+0.070</td><td align="left" valign="top">.046</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MentaLLaMA_Ours vs Joint</td><td align="left" valign="top">0.610</td><td align="left" valign="top">0.404</td><td align="left" valign="top">+0.206</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs LLaMA_Ours</td><td align="left" valign="top">0.676</td><td align="left" valign="top">0.670</td><td align="left" valign="top">+0.006</td><td align="left" valign="top">.91</td></tr><tr><td align="left" valign="top" colspan="5">Last hospitalization</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaMA_Ours vs Direct</td><td align="left" valign="top">0.866</td><td align="left" valign="top">0.754</td><td align="left" valign="top">+0.112</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLaMA_Ours vs Joint</td><td align="left" valign="top">0.866</td><td align="left" valign="top">0.832</td><td align="left" valign="top">+0.034</td><td align="left" valign="top">.28</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs Direct</td><td align="left" valign="top">0.858</td><td align="left" valign="top">0.784</td><td align="left" valign="top">+0.074</td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs Joint</td><td align="left" valign="top">0.858</td><td align="left" valign="top">0.830</td><td align="left" valign="top">+0.028</td><td align="left" valign="top">.44</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>OpenBioLLM_Ours vs Direct</td><td align="left" valign="top">0.868</td><td align="left" valign="top">0.788</td><td align="left" valign="top">+0.080</td><td align="left" valign="top">.006</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>OpenBioLLM_Ours vs Joint</td><td align="left" valign="top">0.868</td><td align="left" valign="top">0.838</td><td align="left" valign="top">+0.030</td><td align="left" valign="top">.31</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MentaLLaMA_Ours vs Direct</td><td align="left" valign="top">0.858</td><td align="left" valign="top">0.742</td><td align="left" valign="top">+0.116</td><td align="left" valign="top">.006</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MentaLLaMA_Ours vs Joint</td><td align="left" valign="top">0.858</td><td align="left" valign="top">0.734</td><td align="left" valign="top">+0.124</td><td align="left" valign="top">.004</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral_Ours vs LLaMA_Ours</td><td align="left" valign="top">0.858</td><td align="left" valign="top">0.866</td><td align="left" valign="top">&#x2212;0.008</td><td align="left" valign="top">.86</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Values represent chart-level accuracy averaged across 10-fold cross-validation. &#x0394; denotes the difference in accuracy between the proposed 2-stage framework (ours) and the comparison method. Statistical significance was assessed using 1000 bootstrap resamples. Lower <italic>P</italic> values indicate stronger evidence that the observed accuracy difference was not due to random variation. </p></fn><fn id="table5fn2"><p><sup>b</sup>Ours: proposed 2-stage framework.</p></fn><fn id="table5fn3"><p><sup>c</sup>Direct: direct chart-level prediction.</p></fn><fn id="table5fn4"><p><sup>d</sup>Joint: single-stage joint extraction.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Results of Additional Analyses</title><sec id="s3-4-1"><title>Oracle Analysis Results</title><p>Oracle analysis revealed heterogeneous effects across models (Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For onset time prediction, oracle inputs modestly improved performance for OpenBioLLM (0.776 vs 0.738) and MentaLLaMA (0.762 vs 0.734), indicating that sentence-level extraction errors contributed to downstream performance limitations in these models. In contrast, oracle performance was comparable to or lower than that of the complete pipeline for LLaMA (0.728 vs 0.744) and Mistral (0.702 vs 0.772). For episode count and hospitalization count, oracle and chain performances were highly similar across all models, suggesting that these tasks were relatively robust to sentence-level extraction errors. For the most recent hospitalization, oracle performance was consistently lower than that of the complete pipeline (eg, Mistral: 0.796 vs 0.858; LLaMA: 0.846 vs 0.866).</p></sec><sec id="s3-4-2"><title>Prompt Sensitivity Analysis Results</title><p>Prompt design had a measurable impact on extraction performance (Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The initial structured prompt produced the lowest overall performance, whereas the simplified plain-text format adopted in the final system achieved the best overall results across most clinical event and temporal extraction tasks. Removing few-shot examples improved performance for several common extraction categories, although performance for persistence and vague temporal expressions declined. While chain-of-thought prompting improved performance on some temporal extraction tasks, it did not consistently improve overall performance. These findings suggest that prompt structure and output format can substantially influence extraction quality in prompt-only evaluations and informed the final prompt design used for subsequent fine-tuning.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>As demonstrated in this study, modern LLMs provide a significant step forward in improving the efficiency and accuracy of the analysis of the course of psychiatric disorder. Our results highlight the potential of LLMs to extract critical clinical information from electronic health records, significantly reducing the manual burden on clinicians while enhancing the quality of the extracted data.</p><p>A notable finding of this study was the consistent advantage of the proposed 2-stage framework over both direct and joint extraction approaches. One possible explanation is that psychiatric discharge summaries often contain multiple temporally distributed events, diagnoses, remissions, and hospitalizations within a single document. Direct prediction requires the model to simultaneously identify relevant events, resolve temporal information, and infer chart-level clinical course features from the entire note, creating a highly complex reasoning task. Although the joint approach explicitly incorporates sentence-level extraction, it still requires event extraction and chart-level prediction to be performed within a single prompt-response process. In contrast, the proposed framework decomposes the task into 2 sequential steps. By first converting free-text narratives into structured event-time representations and then performing chart-level reasoning on the structured outputs, the model can focus on a simpler objective at each stage. This decomposition likely reduces cognitive load, improves interpretability, and facilitates the aggregation of longitudinal clinical information, particularly for first-episode onset prediction, where the largest performance gains were observed. The greatest improvements were observed for first-episode onset prediction, suggesting that separating temporal information extraction from chart-level reasoning is particularly beneficial for reconstructing longitudinal disease trajectories.</p><p>However, oracle analyses suggested that chart-level performance was not determined solely by sentence-level extraction accuracy. Although replacing model-generated sentence-level outputs with gold-standard annotations improved performance for some models and tasks, the improvement was not consistent. This suggests that many chart-level errors arose from the second-stage aggregation and reasoning process, such as selecting the correct onset time, distinguishing episodes from hospitalizations, and integrating multiple temporally distributed events across a discharge summary. In addition, the chart-level extractor was developed using model-generated sentence-level outputs, and gold-standard annotations may differ in format or information density from these predicted inputs. These findings suggest that future improvements should target both sentence-level extraction and chart-level clinical reasoning.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>Our study did not use cloud-based APIs, such as OpenAI&#x2019;s GPT, due to the highly sensitive nature of psychiatric notes, in which data privacy is a critical concern. As a result, we were unable to compare the performance of OpenAI&#x2019;s GPT with other LLMs. Instead, we selected LLaMA and Mistral for this study, as they can be deployed locally and fine-tuned on private data, mitigating privacy risks. While LLaMA and similar models offer greater control and customization, they often require fine-tuning or retrieval augmentation to achieve performance comparable to larger proprietary models such as GPT-4 [<xref ref-type="bibr" rid="ref19">19</xref>]. To address this, researchers have developed domain-specific LLMs for mental health, such as MentaLlama, which integrate psychiatric knowledge through fine-tuning [<xref ref-type="bibr" rid="ref14">14</xref>]. With proper adaptation, open models can significantly reduce the performance gap.</p><p>To further investigate the impact of domain-specific open models, we evaluated the performance of MentaLlama [<xref ref-type="bibr" rid="ref14">14</xref>] and OpenBioLLM [<xref ref-type="bibr" rid="ref15">15</xref>]. However, the results did not show a clear improvement over LLaMA 3.1. A possible explanation is that MentaLlama is based on LLaMA 2, while OpenBioLLM is built on LLaMA 3; their effectiveness may, therefore, be limited by the capabilities of their respective base models.</p><p>Recent studies highlight the potential of Mistral-7B in clinical text extraction. A fine-tuned version of the model achieved 91% accuracy in identifying key clinical history details, closely matching GPT-4&#x2019;s 92%, with no statistically significant difference [<xref ref-type="bibr" rid="ref20">20</xref>]. Despite being an order of magnitude smaller than GPT-4, Mistral-7B, optimized with prompt engineering and in-context learning, reached near-human agreement (&#x03BA;&#x2248;0.75) with radiologists to extract structured history elements.</p><p>These findings suggest that while GPT-4 excels in complex reasoning tasks, smaller open models such as LLaMA and Mistral can be competitive when fine-tuned for domain-specific applications. The choice of model depends on balancing performance, resource availability, and data security. In contrast, LLaMA and Mistral provide greater control through local deployment and fine-tuning, albeit with some trade-offs in raw performance.</p></sec><sec id="s4-3"><title>Error Analysis</title><p>Error analysis revealed several recurring sources of chart-level extraction errors. The most common error category was ambiguous temporal information (55/179, 30.7%), followed by missing information in the source notes (27/183, 14.8%), episode-hospitalization conflation (19/166, 11.4%), and confusion arising from multiple diagnoses with distinct onset times (11/108, 10.2%). Errors related to multiple clinical events described within a single sentence accounted for an additional 8.0% (9/112).</p><p>Missing information in the source notes often limited the accuracy of extraction, regardless of the model&#x2019;s capabilities. For example, in the note &#x201C;Since then, the patient has been repeatedly admitted to psychiatric services due to unstable mood and suicide attempts,&#x201D; the number of hospitalizations was not specified, preventing accurate extraction. Ambiguous temporal information was another major source of error. In one case, a patient was evaluated in the emergency department and subsequently admitted, but no explicit hospitalization date was documented. The model occasionally inferred the outpatient visit date as the admission date, despite the absence of a clearly stated admission date.</p><p>The complexity of psychiatric narratives also contributed to extraction errors. When multiple diagnoses with distinct onset times were described in the same note, the model sometimes failed to correctly associate symptom onset with the appropriate diagnosis. For example, in a note describing autism diagnosed at the age of 4 years and mood symptoms beginning at the age of 15 years, the model occasionally conflated the onset times of the 2 conditions. Similarly, when multiple clinical events were described within a single sentence, the model sometimes captured only one event while overlooking others. For instance, in the sentence &#x201C;Psychotic symptoms improved in March 2010 but worsened in May 2010,&#x201D; the model often identified the symptom exacerbation but failed to recognize the preceding improvement.</p><p>Finally, episode-hospitalization conflation represented a distinct error pattern. In notes describing multiple psychiatric episodes alongside several hospital admissions, the model occasionally interpreted each episode as a separate hospitalization, leading to the overestimation of hospitalization counts. Overall, these findings suggest that many chart-level errors arise from incomplete documentation and the inherent complexity of psychiatric narratives rather than simple failures of event recognition.</p></sec><sec id="s4-4"><title>Limitations</title><p>Despite the promising results, several limitations were identified in this study. First, the dataset used was limited to discharge notes from a single hospital, which may not fully represent the diversity of psychiatric documentation across different clinical settings. Further research is needed to validate the model&#x2019;s performance on larger and more diverse datasets, including outpatient and emergency room records, to ensure broader applicability. Second, the dataset is best characterized as primarily English clinical text with occasional code-mixed elements rather than fully Chinese-language records. Because the evaluated models were primarily pretrained on English corpora and the study used predominantly English discharge summaries, the generalizability of our findings to predominantly Chinese-language psychiatric records or other multilingual clinical settings remains uncertain. Future studies should evaluate the proposed framework on datasets with varying language compositions and across different health care settings. Third, the study does not differentiate between onset times for multiple diagnoses when patients have comorbid psychiatric conditions. This omission can lead to inaccuracies in real-world applications. Fourth, we identified only 4 items related to the clinical course. Expanding the scope of extracted features to include additional relevant clinical indicators&#x2014;such as treatment responses, adverse effects for each regimen, drug compliance assessments, and identification of precipitating factors for relapse and recurrence&#x2014;would provide a more comprehensive view of the patient&#x2019;s psychiatric trajectory. Finally, several risks associated with deploying LLMs in clinical settings warrant explicit consideration. LLMs can generate plausible but incorrect outputs, and the extracted information should therefore be treated as decision-support rather than authoritative clinical data. We emphasize that the present system is a research prototype and must undergo prospective clinical validation before any real-world deployment. All predictions should be reviewed by qualified clinicians before informing treatment decisions.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This study demonstrates the potential of LLMs to automate the extraction of key clinical features from psychiatric discharge summaries. The model effectively captures structured data such as hospitalization events and onset dates, though further refinement is needed to handle vague temporal information and complex clinical nuances. Despite these challenges, LLMs show promise in enhancing psychiatric care by equipping clinicians with efficient, accurate tools for managing psychiatric disorders. As LLM technology advances, its integration into routine practice may improve the quality of care and patient outcomes.</p></sec></sec></body><back><ack><p>The authors thank the staff of the Department of Medical Research, National Taiwan University Hospital for the Integrated Medical Database (NTUH-iMD). Generative AI (GenAI) tools were used during the research and manuscript preparation process under full human supervision. According to the GAIDeT taxonomy (2025), GenAI tools were used for code generation, code optimization, reformatting, publication support, language editing, and manuscript text refinement. The tools used included Claude Sonnet 4.6, ChatGPT, and Grammarly. All AI-assisted outputs were reviewed, verified, and edited by the authors, who take full responsibility for the accuracy, integrity, and scientific content of the final manuscript. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This work was supported by Taiwan&#x2019;s National Science and Technology Council (MOST110-2314-B-400&#x2010;053-MY3, received by CSW) and National Health Research Institutes (CG-113-GP-01, received by CSW). The funding agency had no role in study design, data collection, analysis, publishing decisions, or manuscript preparation.</p></sec><sec><title>Data Availability</title><p>The data are not publicly available due to privacy or ethical restrictions.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: all authors</p><p>Data curation: CHC, CHS</p><p>Formal analysis: CHC, CHS</p><p>Methodology: CHC, CSW, HJD</p><p>Project administration: CSW, YLC</p><p>Resources: CSW, YLC</p><p>Software: CHC, CHS</p><p>Supervision: HJD, HHC</p><p>Validation: HJD, HHC</p><p>Writing &#x2013; original draft: CHC, CSW</p><p>Writing &#x2013; review and editing: all authors</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1"><italic>ICD</italic> </term><def><p><italic>International Classification of Diseases</italic></p></def></def-item><def-item><term id="abb2"><italic>ICD-10-CM</italic></term><def><p><italic>International Classification of Diseases, Tenth Revision, Clinical Modification</italic></p></def></def-item><def-item><term id="abb3"><italic>ICD-9-CM</italic></term><def><p><italic>International Classification of Diseases, Ninth Revision, Clinical Modification</italic></p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">LoRA</term><def><p>low-rank adaptation</p></def></def-item><def-item><term id="abb6">MAE</term><def><p>mean absolute error</p></def></def-item><def-item><term id="abb7">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb8">NTUH-iMD</term><def><p>Integrated Medical Database of National Taiwan University Hospital</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Emsley</surname><given-names>R</given-names> </name><name name-style="western"><surname>Chiliza</surname><given-names>B</given-names> </name><name name-style="western"><surname>Asmal</surname><given-names>L</given-names> </name><name name-style="western"><surname>Harvey</surname><given-names>BH</given-names> </name></person-group><article-title>The nature of relapse in schizophrenia</article-title><source>BMC Psychiatry</source><year>2013</year><month>02</month><day>8</day><volume>13</volume><fpage>50</fpage><pub-id pub-id-type="doi">10.1186/1471-244X-13-50</pub-id><pub-id pub-id-type="medline">23394123</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Angst</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sellaro</surname><given-names>R</given-names> </name></person-group><article-title>Historical perspectives and natural history of bipolar disorder</article-title><source>Biol Psychiatry</source><year>2000</year><month>09</month><day>15</day><volume>48</volume><issue>6</issue><fpage>445</fpage><lpage>457</lpage><pub-id pub-id-type="doi">10.1016/s0006-3223(00)00909-4</pub-id><pub-id pub-id-type="medline">11018218</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Treuer</surname><given-names>T</given-names> </name><name name-style="western"><surname>Tohen</surname><given-names>M</given-names> </name></person-group><article-title>Predicting the course and outcome of bipolar disorder: a review</article-title><source>Eur Psychiatry</source><year>2010</year><month>10</month><volume>25</volume><issue>6</issue><fpage>328</fpage><lpage>333</lpage><pub-id pub-id-type="doi">10.1016/j.eurpsy.2009.11.012</pub-id><pub-id pub-id-type="medline">20444581</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Velupillai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Suominen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liakata</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Using clinical natural language processing for health outcomes research: overview and actionable suggestions for future advances</article-title><source>J Biomed Inform</source><year>2018</year><month>12</month><volume>88</volume><fpage>11</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2018.10.005</pub-id><pub-id pub-id-type="medline">30368002</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Viani</surname><given-names>N</given-names> </name><name name-style="western"><surname>Botelle</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kerwin</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A natural language processing approach for identifying temporal disease onset information from mental healthcare text</article-title><source>Sci Rep</source><year>2021</year><month>01</month><day>12</day><volume>11</volume><issue>1</issue><fpage>757</fpage><pub-id pub-id-type="doi">10.1038/s41598-020-80457-0</pub-id><pub-id pub-id-type="medline">33436814</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Agrawal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hegselmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sontag</surname><given-names>D</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Goldberg</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kozareva</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name></person-group><article-title>Large language models are few-shot clinical information extractors</article-title><source>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</source><year>2022</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>1998</fpage><lpage>2022</lpage><pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main.130</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Van Veen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Van Uden</surname><given-names>C</given-names> </name><name name-style="western"><surname>Blankemeier</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Adapted large language models can outperform medical experts in clinical text summarization</article-title><source>Nat Med</source><year>2024</year><month>04</month><volume>30</volume><issue>4</issue><fpage>1134</fpage><lpage>1142</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-02855-5</pub-id><pub-id pub-id-type="medline">38413730</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Mahbub</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dams</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Srinivasan</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Leveraging large language models to extract information on substance use disorder severity from clinical notes: a zero-shot learning approach</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 18, 2024</comment><pub-id pub-id-type="doi">10.1038/s44184-024-00114-6</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Su</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Chien</surname><given-names>YL</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>HH</given-names> </name></person-group><article-title>Augmenting DSM-5 diagnostic criteria with self-attention-based BiLSTM models for psychiatric diagnosis</article-title><source>Artif Intell Med</source><year>2023</year><month>02</month><volume>136</volume><fpage>102488</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2023.102488</pub-id><pub-id pub-id-type="medline">36710066</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>YQ</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>CT</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>CC</given-names> </name><etal/></person-group><article-title>Unlocking the secrets behind advanced artificial intelligence language models in deidentifying Chinese-English mixed clinical text: development and validation study</article-title><source>J Med Internet Res</source><year>2024</year><month>01</month><day>25</day><volume>26</volume><fpage>e48443</fpage><pub-id pub-id-type="doi">10.2196/48443</pub-id><pub-id pub-id-type="medline">38271060</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>CK</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>FD</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>YQ</given-names> </name><etal/></person-group><article-title>Principle-based approach for the de-identification of code-mixed electronic health records</article-title><source>IEEE Access</source><year>2022</year><volume>10</volume><fpage>22875</fpage><lpage>22885</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2022.3148396</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Touvron</surname><given-names>H</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Stone</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Llama 2: open foundation and fine-tuned chat models</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 18, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2307.09288</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kuang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name></person-group><article-title>MentaLLaMA: interpretable mental health analysis on social media with large language models</article-title><conf-name>WWW &#x2019;24: The ACM Web Conference 2024</conf-name><conf-date>May 13-17, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3589334.3648137</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>OpenBioLLM: advancing open-source large language models for healthcare and life sciences</article-title><source>Hugging Face</source><year>2024</year><access-date>2026-07-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/aaditya/OpenBioLLM-Llama3-70B">https://huggingface.co/aaditya/OpenBioLLM-Llama3-70B</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><article-title>Mistralai/Ministral-8B-Instruct-2410</article-title><source>Hugging Face</source><year>2024</year><access-date>2026-07-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/mistralai/Ministral-8B-Instruct-2410">https://huggingface.co/mistralai/Ministral-8B-Instruct-2410</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Alsentzer</surname><given-names>E</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boag</surname><given-names>W</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Jindi</surname><given-names>D</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name></person-group><article-title>Publicly available clinical BERT embeddings</article-title><source>Proceedings of the 2nd Clinical Natural Language Processing Workshop, Pages 72&#x2013;78, Minneapolis, Minnesota, USA</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>72</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.18653/v1/W19-1909</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wallis</surname><given-names>P</given-names> </name><name name-style="western"><surname>Allen-Zhu</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>LoRA: low-rank adaptation of large language models</article-title><access-date>2026-07-23</access-date><conf-name>Tenth International Conference on Learning Representations (ICLR 2022)</conf-name><conf-date>Apr 25-29, 2022</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=nZeVKeeFYf9">https://openreview.net/pdf?id=nZeVKeeFYf9</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Boggavarapu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Srivastava</surname><given-names>V</given-names> </name><name name-style="western"><surname>Varanasi</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bhaumik</surname><given-names>R</given-names> </name></person-group><article-title>Evaluating enhanced LLMs for precise mental health diagnosis from clinical notes</article-title><source>medRxiv</source><comment>Preprint posted online on  Dec 17, 2024</comment><pub-id pub-id-type="doi">10.1101/2024.12.16.24317648</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Larson</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Koirala</surname><given-names>A</given-names> </name><name name-style="western"><surname>Cheuy</surname><given-names>LY</given-names> </name><etal/></person-group><article-title>Assessing completeness of clinical histories accompanying imaging orders using adapted open-source and closed-source large language models</article-title><source>Radiology</source><year>2025</year><month>02</month><volume>314</volume><issue>2</issue><fpage>e241051</fpage><pub-id pub-id-type="doi">10.1148/radiol.241051</pub-id><pub-id pub-id-type="medline">39998369</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary tables supporting the main analyses.</p><media xlink:href="formative_v10i1e94454_app1.docx" xlink:title="DOCX File, 41 KB"/></supplementary-material></app-group></back></article>