<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e84974</article-id><article-id pub-id-type="doi">10.2196/84974</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Fine-Tuning Large Language Models for Structured Extraction of Infectious Disease&#x2013;Related Information From Clinical Notes in Japanese Primary Care: Development and Internal Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Yoshihara</surname><given-names>Hiroshi</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Maeda</surname><given-names>Haruka</given-names></name><degrees>MD, MPH, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hagiwara</surname><given-names>Yuriko</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sato</surname><given-names>Daichi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kitajima</surname><given-names>Kei</given-names></name><degrees>MEng</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Iwata</surname><given-names>Akihiro</given-names></name><degrees>MIA</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Van de Velde</surname><given-names>Nicolas</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Nakamura</surname><given-names>Yuta</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yamagishi</surname><given-names>Yosuke</given-names></name><degrees>MSc, MD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Igarashi</surname><given-names>Ataru</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Health Policy and Public Health, Graduate School of Pharmaceutical Sciences, The University of Tokyo</institution><addr-line>7-3-1 Hongo, Bunkyo-ku</addr-line><addr-line>Tokyo</addr-line><country>Japan</country></aff><aff id="aff2"><institution>Department of Respiratory Infections, Institute of Tropical Medicine, Nagasaki University</institution><addr-line>Nagasaki</addr-line><country>Japan</country></aff><aff id="aff3"><institution>M3 Inc.</institution><addr-line>Tokyo</addr-line><country>Japan</country></aff><aff id="aff4"><institution>Moderna, Inc.</institution><addr-line>Cambridge</addr-line><addr-line>MA</addr-line><country>United States</country></aff><aff id="aff5"><institution>Department of Computational Diagnostic Radiology and Preventive Medicine, Graduate School of Medicine, The University of Tokyo</institution><addr-line>Tokyo</addr-line><country>Japan</country></aff><aff id="aff6"><institution>Division of Radiology and Biomedical Engineering, Graduate School of Medicine, The University of Tokyo</institution><addr-line>Tokyo</addr-line><country>Japan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Miyawaki</surname><given-names>Atsushi</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Babar</surname><given-names>Prasad</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Elsanousi</surname><given-names>Yasir</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Hiroshi Yoshihara, MSc, Department of Health Policy and Public Health, Graduate School of Pharmaceutical Sciences, The University of Tokyo, 7-3-1 Hongo, Bunkyo-ku, Tokyo, 113-0033, Japan, 81 3-5841-4828; <email>no14nec0@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>9</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e84974</elocation-id><history><date date-type="received"><day>28</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>20</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>21</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Hiroshi Yoshihara, Haruka Maeda, Yuriko Hagiwara, Daichi Sato, Kei Kitajima, Akihiro Iwata, Nicolas Van de Velde, Yuta Nakamura, Yosuke Yamagishi, Ataru Igarashi. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 10.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e84974"/><abstract><sec><title>Background</title><p>The COVID-19 pandemic highlighted the importance of timely infectious disease surveillance. In Japan, conventional sentinel and claims-based systems incur reporting lags and capture limited clinical detail, whereas free-text clinical notes in electronic health records (EHRs) hold richer, timelier symptom and vaccination information. Natural language processing (NLP) with large language models (LLMs) offers a way to structure such free text at scale.</p></sec><sec><title>Objective</title><p>We aimed to develop and internally validate an NLP algorithm to extract structured infectious disease&#x2013;related symptoms and vaccination history from free-text clinical notes in Japanese primary care, as a feasibility step toward low-latency, EHR-based surveillance.</p></sec><sec sec-type="methods"><title>Methods</title><p>A total of 773 clinical notes, originating from 526 unique patients, were provided by M3 Inc through the Japan Medical Data Survey and used for analysis. Three physicians annotated information related to infectious disease symptoms and vaccination history. The data were divided into 622 (80%) training cases and 151 (20%) evaluation cases with no patient overlap. We compared a physician-designed, rule-based algorithm, few-shot learning (FSL) using commercial and open-source LLMs, and supervised fine-tuning (SFT) of open-source LLMs, using the macroaveraged <italic>F</italic><sub>1</sub>-score (unweighted mean across 9 clinical categories). Sensitivity, specificity, positive predictive value (PPV), and negative predictive value (NPV) were also computed, with 95% CIs from a patient-level cluster bootstrap (2000 replicates).</p></sec><sec sec-type="results"><title>Results</title><p>Rule-based extraction achieved a macroaveraged <italic>F</italic><sub>1</sub>-score of 0.685 (95% CI 0.630-0.736). FSL markedly improved the extraction of high-variability items such as vaccination history and onset date. Anthropic Claude 3.5 Sonnet achieved a macroaveraged <italic>F</italic><sub>1</sub>-score of 0.875 (95% CI 0.800-0.913; sensitivity 0.929, specificity 0.918). SFT of Google&#x2019;s open-source Gemma 2 27B model with quantized low-rank adaptation (QLoRA) achieved the highest point estimate (macroaveraged <italic>F</italic><sub>1</sub>-score of 0.906, 95% CI 0.833-0.945; sensitivity 0.921, specificity 0.969, PPV 0.906); the difference from Claude 3.5 Sonnet was small and not statistically distinguishable (&#x0394;<italic>F</italic><sub>1</sub>-score=0.030, 95% CI &#x2212;0.035 to 0.140). A small, fine-tuned Gemma 2 2B model reached 0.822 (95% CI 0.752-0.874), significantly lower than that of the 27B model (&#x0394;<italic>F</italic><sub>1</sub>-score=0.084, 95% CI 0.040-0.163).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>A fine-tuned open-source LLM can accurately extract and structure infectious disease&#x2013;related information from Japanese free-text clinical notes, achieving performance comparable to that of a commercial model while enabling processing within a closed environment. These findings support the feasibility of EHR-based digital surveillance, whose downstream utility remains to be demonstrated.</p></sec></abstract><kwd-group><kwd>natural language processing</kwd><kwd>large language model</kwd><kwd>electronic health record</kwd><kwd>EHR</kwd><kwd>infectious diseases</kwd><kwd>public health</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The COVID-19 pandemic underscored the importance of timely infectious disease surveillance, and this need has persisted in the postpandemic era. Influenza, COVID-19, and respiratory syncytial virus (RSV) now cocirculate, present overlapping symptoms, and can cause simultaneous outbreaks that strain health care systems [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Surveillance systems that can rapidly detect and characterize such activity therefore remain a public health priority.</p><p>In Japan, harnessing electronic health record (EHR) data has the potential to offer particular advantages for infectious disease surveillance over traditional claims data. The country&#x2019;s conventional surveillance systems, such as the national sentinel provider network for influenza, typically aggregate case reports on a weekly basis and often incur reporting lags of approximately 1 week [<xref ref-type="bibr" rid="ref4">4</xref>]. This delay can slow down outbreak detection. To achieve lower latency, researchers have turned to alternative data sources such as health insurance claims and pharmacy records. Notably, a real-time prescription surveillance system introduced in Japan demonstrated that daily tracking of antiviral drug prescriptions could closely estimate influenza case numbers in near real time [<xref ref-type="bibr" rid="ref4">4</xref>]. However, although claims and prescription data provide faster updates, they contain only limited clinical details, primarily billing codes for diagnoses or medications, and may not capture the nuanced symptom information available in a clinical note. Claims-based metrics also risk being biased by changing coding practices and may lack clinical fidelity [<xref ref-type="bibr" rid="ref5">5</xref>]. By contrast, EHR clinical notes contain rich descriptions of patient symptoms, findings, and vaccination history recorded at the point of care. These narratives can be accessed soon after the medical encounter, giving a timelier and more detailed picture of emerging infections. Studies have found that surveillance based on EHR-derived clinical data can provide more objective and reliable estimates of disease incidence than surveillance based on claims data [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>Natural language processing (NLP) and deep learning are promising tools to address some of these surveillance challenges. NLP is a branch of AI that focuses on enabling computers to understand and generate human language [<xref ref-type="bibr" rid="ref6">6</xref>]. Recently, deep learning techniques have advanced NLP capabilities, allowing automated analysis of large volumes of text. This is especially significant in health care, where enormous amounts of data are recorded as unstructured text in EHRs [<xref ref-type="bibr" rid="ref6">6</xref>]. Applications of NLP in health care are broad: systems have been developed to classify medical notes by content, recognize and extract clinical entities (such as symptoms, diagnoses, or medications) from text, summarize patient records, and even translate medical jargon into lay language [<xref ref-type="bibr" rid="ref7">7</xref>]. Recent large language models (LLMs) are used to parse free-text EHR narratives and extract structured clinical concepts with unprecedented accuracy. Many studies have concluded that commercial and open-source LLMs could outperform traditional NLP systems in extracting various types of information from medical notes [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Internationally, numerous studies in the Netherlands [<xref ref-type="bibr" rid="ref10">10</xref>], the United States [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>], Singapore [<xref ref-type="bibr" rid="ref14">14</xref>], and New Zealand [<xref ref-type="bibr" rid="ref15">15</xref>] have used NLP to identify infectious disease cases or to extract related clinical information from EHRs for case detection and surveillance. In Japan, however, epidemiologists have relied mainly on structured data, such as outpatient prescription records [<xref ref-type="bibr" rid="ref16">16</xref>], and, to our knowledge, no study has directly extracted infectious disease information from unstructured Japanese clinical narratives. In this study, we aim to address this gap by developing and validating an NLP algorithm that leverages state-of-the-art LLMs to extract infectious disease&#x2013;related symptoms and vaccination history from free-text clinical notes written in Japanese, as a feasibility step toward timelier, EHR-based digital surveillance.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Data Source and Study Population</title><p>A pseudonymized extract of clinical notes from the Japan Medical Data Survey (JAMDAS; M3 Inc), an outpatient health care database of nationwide primary care clinics curated by M3 Inc [<xref ref-type="bibr" rid="ref17">17</xref>], was used for analysis. The 5 outpatient clinics were a convenience sample of facilities participating in the database; they were not randomly drawn from JAMDAS, and patients within these clinics were randomly sampled with no filtering by disease name or prescription. The dataset therefore includes patients with a variety of medical conditions. It comprised 2020 clinical notes from 1036 patients aged 0 to 84 years who visited these clinics for fever or suspected infectious disease between October 2020 and September 2023. The clinics served distinct populations (median patient age ranging from approximately 1 year at a pediatric-predominant clinic to approximately 37 years at an adult clinic; Table S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-2"><title>Annotation and Interrater Agreement</title><p>After removing missing values and duplicate records, 3 physicians within the research team performed annotations on the following items: body temperature, fatigue, muscle and joint pain, headache and dizziness, respiratory symptoms, visible changes in the throat, gastrointestinal symptoms, sensory abnormalities, other symptoms, vaccination information, and onset date. Detailed annotation items, criteria, and extraction formats are described in <xref ref-type="table" rid="table1">Table 1</xref>. The raw onset description, parsed onset date, body temperature, and description of vaccination history were annotated as text data types, while all other items were classified as binary variables.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Dataset characteristics and extraction item definitions<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Items</td><td align="left" valign="bottom">Training dataset (n=622)</td><td align="left" valign="bottom">Evaluation dataset (n=151)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Definition</td><td align="left" valign="bottom">Format</td><td align="left" valign="bottom">Example</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="7">Demographic information</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age (years), mean (SD)</td><td align="left" valign="top">16.2 (15.7)</td><td align="left" valign="top">18.3 (16.8)</td><td align="left" valign="top">.16</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male sex, n (%)</td><td align="left" valign="top">350 (56.3)</td><td align="left" valign="top">70 (46.4)</td><td align="left" valign="top">.04</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top" colspan="7">Symptom onset, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Parsed date mentioned</td><td align="left" valign="top">496 (79.7)</td><td align="left" valign="top">119 (78.8)</td><td align="left" valign="top">.89</td><td align="left" valign="top">Date of initial symptom onset determined from the visit date and clinical note description</td><td align="left" valign="top">Text</td><td align="left" valign="top">2022-10-15</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Raw description mentioned</td><td align="left" valign="top">503 (80.9)</td><td align="left" valign="top">121 (80.1)</td><td align="left" valign="top">.93</td><td align="left" valign="top">Raw description of the initial symptom onset date in the clinical note</td><td align="left" valign="top">Text</td><td align="left" valign="top">&#x6628;&#x65E5;&#x671D; (English: yesterday morning)</td></tr><tr><td align="left" valign="top" colspan="7">Body temperature, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Body temperature mentioned</td><td align="left" valign="top">420 (67.5)</td><td align="left" valign="top">99 (65.6)</td><td align="left" valign="top">.72</td><td align="left" valign="top">Highest body temperature recorded in the clinical note. Annotations were standardized to the format &#x201C;3x.x&#x201D;</td><td align="left" valign="top">Text</td><td align="left" valign="top">39.4</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Slight fever</td><td align="left" valign="top">78 (12.5)</td><td align="left" valign="top">18 (11.9)</td><td align="left" valign="top">.94</td><td align="left" valign="top">Whether slight fever is mentioned in the clinical note or the patient&#x2019;s highest body temperature is between 37 &#x00B0;C and 37.4 &#x00B0;C</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Fever</td><td align="left" valign="top">369 (59.3)</td><td align="left" valign="top">89 (58.9)</td><td align="left" valign="top">1</td><td align="left" valign="top">Whether fever is mentioned in the clinical note or the patient&#x2019;s highest body temperature is &#x2265;37.0 &#x00B0;C</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High fever</td><td align="left" valign="top">190 (30.5)</td><td align="left" valign="top">47 (31.1)</td><td align="left" valign="top">.97</td><td align="left" valign="top">Whether high fever is mentioned in the clinical note or patient&#x2019;s highest body temperature is &#x2265;38 &#x00B0;C</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Chills</td><td align="left" valign="top">12 (1.9)</td><td align="left" valign="top">3 (2)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="7">Fatigue and muscle or joint pain, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Fatigue</td><td align="left" valign="top">89 (14.3)</td><td align="left" valign="top">21 (13.9)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Muscle or joint pain</td><td align="left" valign="top">18 (2.9)</td><td align="left" valign="top">5 (3.3)</td><td align="left" valign="top">.997</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="7">Headache and dizziness, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Dizziness</td><td align="left" valign="top">3 (0.5)</td><td align="left" valign="top">1 (0.7)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Headache</td><td align="left" valign="top">139 (22.3)</td><td align="left" valign="top">33 (21.9)</td><td align="left" valign="top">.98</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="7">Respiratory symptoms, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Runny nose</td><td align="left" valign="top">222 (35.7)</td><td align="left" valign="top">54 (35.8)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Nasal congestion</td><td align="left" valign="top">17 (2.7)</td><td align="left" valign="top">4 (2.6)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sore throat</td><td align="left" valign="top">239 (38.4)</td><td align="left" valign="top">57 (37.7)</td><td align="left" valign="top">.95</td><td align="left" valign="top">Including discomfort in the throat</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cough</td><td align="left" valign="top">275 (44.2)</td><td align="left" valign="top">68 (45)</td><td align="left" valign="top">.93</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sputum</td><td align="left" valign="top">104 (16.7)</td><td align="left" valign="top">24 (15.9)</td><td align="left" valign="top">.90</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Chest pain</td><td align="left" valign="top">3 (0.5)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">.90</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Shortness of breath</td><td align="left" valign="top">3 (0.5)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">.90</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Breathing sounds</td><td align="left" valign="top">11 (1.8)</td><td align="left" valign="top">3 (2)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of abnormal breath sounds based solely on objective physical examination findings (excluding patient-reported symptoms)</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="7">Throat appearance, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Tonsillar hypertrophy</td><td align="left" valign="top">1 (0.2)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Tonsillar exudates</td><td align="left" valign="top">3 (0.5)</td><td align="left" valign="top">1 (0.7)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pharyngeal redness</td><td align="left" valign="top">113 (18.2)</td><td align="left" valign="top">27 (17.9)</td><td align="left" valign="top">1</td><td align="left" valign="top">Including redness of the tonsils</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top" colspan="7">Gastrointestinal symptoms, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Abdominal pain</td><td align="left" valign="top">22 (3.5)</td><td align="left" valign="top">5 (3.3)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Diarrhea</td><td align="left" valign="top">28 (4.5)</td><td align="left" valign="top">7 (4.6)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Nausea</td><td align="left" valign="top">21 (3.4)</td><td align="left" valign="top">5 (3.3)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Vomiting</td><td align="left" valign="top">21 (3.4)</td><td align="left" valign="top">5 (3.3)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="7">Other symptoms, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Taste abnormality</td><td align="left" valign="top">4 (0.6)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">.72</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Smell abnormality</td><td align="left" valign="top">1 (0.2)</td><td align="left" valign="top">1 (0.7)</td><td align="left" valign="top">.84</td><td align="left" valign="top">Presence of the symptom</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Lymph node swelling</td><td align="left" valign="top">3 (0.5)</td><td align="left" valign="top">1 (0.7)</td><td align="left" valign="top">1</td><td align="left" valign="top">Including swelling of all lymph nodes, regardless of location</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rash</td><td align="left" valign="top">11 (1.8)</td><td align="left" valign="top">3 (2)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of rash or skin eruption</td><td align="left" valign="top">Binary</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top" colspan="7">Vaccination information, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>COVID-19 vaccination</td><td align="left" valign="top">141 (22.7)</td><td align="left" valign="top">34 (22.5)</td><td align="left" valign="top">1</td><td align="left" valign="top">Presence of COVID-19 vaccination, regardless of the number of doses</td><td align="left" valign="top">Binary</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Raw description mentioned</td><td align="left" valign="top">194 (31.2)</td><td align="left" valign="top">47 (31.1)</td><td align="left" valign="top">1</td><td align="left" valign="top">Descriptions related to vaccines (including those other than COVID-19)</td><td align="left" valign="top">Text</td><td align="left" valign="top">&#x30B3;&#x30ED;&#x30CA;&#x30EF;&#x30AF;&#x30C1;&#x30F3;&#x63A5;&#x7A2E;&#xFF1A;2&#x56DE;, &#x6700;&#x7D42;2021&#x5E74;10&#x6708;16&#x65E5;,Pfizer (English: COVID-19 vaccination: 2 doses, most recent on October 16, 2021, Pfizer)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Data are from a development and internal validation study of large language models for extracting infectious disease&#x2013;related information from free-text clinical notes of Japanese primary care outpatients in the Japan Medical Data Survey database between October 2020 and September 2023. The table shows patient demographics and the distribution of each extraction item in the training (n=622) and evaluation (n=151) datasets, together with each item&#x2019;s annotation criteria, extraction format, and worked examples. Binary items are summarized as n (%) notes in which the item was mentioned, and continuous variables (eg, body temperature) are summarized by their distribution. <italic>P</italic> values compare the training and evaluation sets.</p></fn><fn id="table1fn2"><p><sup>b</sup>Not applicable. </p></fn></table-wrap-foot></table-wrap><p>To assess interrater reliability, a pilot set of 216 notes was independently annotated by all 3 physicians; the remaining notes were annotated by a single physician. In the pilot set, the mean observed agreement for binary symptom flags was 93.9% (unanimous, 3/3, 91.1%), with a pooled Fleiss &#x03BA; of 0.31; the low &#x03BA; reflects the very low prevalence of most symptoms (the &#x03BA; paradox) rather than poor agreement. For the fever flag, the physicians&#x2019; operational definitions initially diverged when no temperature had been measured (&#x03BA;&#x2248;0.02); this was resolved by consensus into a single harmonized definition that was applied to the finalized reference standard. For free-text and numeric items, exact-match agreement among all 3 physicians was 71.8% for body temperature and 99.5% for vaccination descriptions (per-item Fleiss &#x03BA; and observed agreement are given in Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Disagreements were resolved by majority vote (&#x2265;2 of 3 physicians). Of the 151 evaluation notes, 45 (30%) were triple annotated and 106 (70%) were single annotated; a sensitivity analysis showed comparable extraction performance between the 2 subsets (Table S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>Owing to the cost of manual annotation, only a portion of the 2020 candidate notes could be annotated: 802 (40%) notes (526 patients) were annotated, and 773 (96%) of these, after removing notes with missing or duplicated diagnostic free text, formed the analysis set. The remaining unannotated candidate notes were therefore excluded because of labeling resource constraints and not because the annotators failed to reach consensus.</p></sec><sec id="s2-3"><title>Algorithm Development</title><p>The dataset was randomly divided into 622 (80%) training cases and 151 (20%) evaluation cases, ensuring no patient overlap between the two sets. We developed and compared the following approaches: a rule-based algorithm crafted by physicians; few-shot learning (FSL) using commercial and open-source LLMs; and fine-tuning of open-source LLMs. The rule-based algorithm was developed by physicians who carefully reviewed the training dataset and translated patterns for each extraction item into regular expressions.</p><p>After several preliminary performance experiments, this study selected Claude 3.5 Sonnet (version 20241022; Anthropic PBC) [<xref ref-type="bibr" rid="ref18">18</xref>] as the representative commercial LLM and the Gemma 2 family (27B and 2B models; Alphabet LLC) [<xref ref-type="bibr" rid="ref19">19</xref>] as the representative open-source LLMs. We tested several prompt patterns and settled on one that outputs results in a JSON format, where each extraction item serves as a key and the corresponding extracted value serves as its value (eg<italic>, {&#x201C;fever&#x201D;: 1, &#x201C;temperature&#x201D;: 38.3, &#x201C;headache&#x201D;: 0}</italic>). The prompt used is shown in <xref ref-type="other" rid="box1">Textbox 1</xref>.</p><boxed-text id="box1"><title> Large language model prompts used for structured data extraction.</title><p><bold>Prompt for few-shot learning</bold></p><p>System: Carefully analyze the following instruction, drawing upon your extensive knowledge and relevant references.</p><p>Keep the answer short and do not provide explanations or notes.</p><p/><p/><p># Instructions:</p><p>Your task is to extract items from a medical record in the context and return the results as JSON strings.</p><p>If you don&#x2019;t find an item in the context or you are not sure, skip the item.</p><p/><p/><p># Format:</p><p>Return JSON strings ONLY in a single line.</p><p>Output a single JSON with multiple keys.</p><p>JSON key must be one of the items below, do NOT change the item name.</p><p>JSON value is the extracted number or text.</p><p>Remove units such as kg.</p><p/><p/><p># Items:</p><p>{description of extraction items}</p><p>For binary items, negative statement = =0, positive statement = =1.</p><p/><p/><p># Examples:</p><p>{multiple examples}</p><p/><p/><p># Now please extract items from the context below.</p><p>Context: {clinical note to process}</p><p>Answer:</p><p/><p/><p><bold>Prompt for fine-tuning</bold></p><p/><p/><p># Instructions:</p><p>Your task is to extract items from a medical record in the context and return the results as JSON strings.</p><p>If you don&#x2019;t find an item in the context or you are not sure, skip the item.</p><p/><p/><p># Format:</p><p>Return JSON strings ONLY in a single line.</p><p>Output a single JSON with multiple keys.</p><p>JSON key must be one of the items below, do NOT change the item name.</p><p>Do NOT change the order of items.</p><p>JSON value is the extracted number or text.</p><p>Remove units such as kg.</p><p/><p/><p># Items:</p><p>{description of extraction items}</p><p>For binary items, negative statement = 0, positive statement = 1.</p><p/><p/><p># Examples:</p><p>{single example}</p><p/><p/><p># Now please extract items from the context below.</p><p>Context:{clinical note to process}</p><p>Answer:</p></boxed-text><p>Within each of 9 clinical categories, per-note item observations were pooled (microaveraged) to compute <italic>F</italic><sub>1</sub>-score; the overall value is the unweighted macroaverage across categories. Each cell shows the <italic>F</italic><sub>1</sub>-score point estimate with its 95% CI from a patient-level cluster bootstrap (2000 replicates). Categories with no positive cases were undefined and excluded from the macroaverage.</p><p>FSL is a technique in which a model is given a small number of example inputs and outputs to learn from before performing a task [<xref ref-type="bibr" rid="ref20">20</xref>]. This allows the model to better understand the task and generate more accurate results without updating its parameters. We initially planned to compare zero-shot inference with FSL inference. However, we found that zero-shot outputs often did not adhere strictly to the JSON format, making reliable evaluation difficult. Therefore, we used the FSL results as the baseline for LLM performance. We randomly sampled 18 examples, ensuring that each extraction item had at least 1 positive instance included.</p><p>To further improve performance, we applied supervised fine-tuning to the open-source LLMs. The training dataset was split into 500 (80%) samples for training and 122 (20%) samples for validation, with the validation set used to monitor overfitting during training. We performed fine-tuning using the instruction-tuned Google Gemma 2 models (gemma-2-2b-it and gemma-2-27b-it). Full fine-tuning was conducted on the 2B model, while, owing to hardware limitations, we applied quantized low-rank adaptation (QLoRA) for fine-tuning the 27B model. Low-rank adaptation (LoRA) is a technique that reduces the number of trainable parameters by injecting low-rank matrices into a pretrained model&#x2019;s weights, enabling efficient fine-tuning without updating the entire model [<xref ref-type="bibr" rid="ref21">21</xref>]. QLoRA combines LoRA with quantization, compressing model weights to reduce memory use while maintaining accuracy, allowing large models to be fine-tuned on limited hardware resources [<xref ref-type="bibr" rid="ref22">22</xref>]. The specific training parameters are detailed in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Fine-tuning was performed using the transformers [<xref ref-type="bibr" rid="ref23">23</xref>] and trl [<xref ref-type="bibr" rid="ref24">24</xref>] libraries on a single RTX 6000 Ada GPU (NVIDIA Corp).</p></sec><sec id="s2-4"><title>Evaluation</title><p>We grouped the extraction items into 9 clinical categories and computed performance within each category. Each extracted item in a note was treated as 1 binary observation; for items extracted as text or numeric values (eg, body temperature, onset date, and vaccination description), an exact-value match was required. Within each category, per-note item observations were pooled (microaveraging) to compute the <italic>F</italic><sub>1</sub>-score, sensitivity, specificity, positive predictive value (PPV), and negative predictive value (NPV); the overall value for each metric was the unweighted macroaverage across the 9 categories, which served as the primary performance metric. Categories with no positive cases in the evaluation set were undefined and were excluded from the average. Metrics were thus computed at the level of item presence within a note, pooled within a category and then macroaveraged across categories&#x2014;not aggregated at the patient or note level. Moreover, 95% CIs were obtained by patient-level cluster bootstrap resampling with 2000 replicates; and paired between-model differences were assessed with the same patient-level bootstrap. Per-category sensitivity, specificity, PPV, and NPV are provided in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-5"><title>Ethical Considerations</title><p>This study was approved by the Nihonbashi-Sakura Clinic Ethics Committee. The committee did not issue an approval or case number for this study, so none is reported. Because the analysis used pseudonymized, retrospective data recorded during routine clinical care, individual informed consent was not obtained; instead, in accordance with the Japanese Ethical Guidelines for Medical and Biological Research Involving Human Subjects, information about the study was disclosed to patients and an opt-out opportunity to decline participation was provided. All records were pseudonymized before they were made available to the research team, and the data were accessed and handled solely in accordance with the restricted-use data agreement with M3 Inc that governs the JAMDAS database. The study was conducted in accordance with the Declaration of Helsinki.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>Of the 2020 candidate notes (1036 patients), 802 (526 patients) were annotated; the remainder were left unannotated owing to labeling resource constraints. After removing notes with missing or duplicated diagnostic free text, 773 notes from 526 patients formed the analysis set. Compared with the annotated notes, the unannotated notes came from patients who were significantly younger (median age 3, IQR 2-17 vs 15, IQR 7-20 years; <italic>P</italic>&#x003C;.001) and were drawn disproportionately from particular clinics (<italic>P</italic>&#x003C;.001); sex did not differ (<italic>P</italic>=.44; Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The data flow and an overview of the experiments are shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. Background information and the distribution of extraction items for the training and evaluation datasets are presented in <xref ref-type="table" rid="table1">Table 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Data flowchart and study overview. LLM: large language model; SFT: supervised fine-tuning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e84974_fig01.png"/></fig><p>The diagram shows the flow from 2020 candidate notes (1036 patients) through annotation (802 notes; 526 patients) to the analysis set (773 notes; 526 patients), and the random, patient-disjoint split into a training set (622 notes; 500 training and 122 validation) and a held-out evaluation set (151 notes) used to compare the rule-based, few-shot, and fine-tuned approaches.</p><p>Interrater reliability was assessed on a 216-note pilot set independently annotated by all 3 physicians; the remaining notes, including 106 (70%) of the 151 evaluation notes, were single annotated.</p><p>As shown in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, although there were some systematic differences in clinical note templates across clinics and physician preferences within clinics, the notes were generally written in the SOAP (subjective, objective, assessment, and plan) format, showing overall similarity. The performance evaluation results for each model are presented in <xref ref-type="table" rid="table2">Table 2</xref>. Some extraction items exhibited relatively low variability in their descriptions, indicating that their descriptions were similar across many physicians; consequently, the rule-based algorithm achieved high performance on these items. For items with high variability in descriptions, such as vaccination information and onset dates, the rule-based algorithm showed low accuracy; however, FSL with LLMs achieved high accuracy. In particular, the commercial model Anthropic Claude 3.5 Sonnet achieved a macroaverage <italic>F</italic><sub>1</sub>-score of 0.875 (95% CI 0.800-0.913), with a sensitivity of 0.929.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Extraction performance by category and algorithm<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Categories</td><td align="left" valign="bottom">Rule-based, <italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Gemma 2 2B+FSL<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>, <italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Gemma 2 27B+FSL, <italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Claude 3.5 Sonnet+FSL, <italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Gemma 2 2B+SFT<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>, <italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Gemma 2 27B+quantized low-rank adaptation SFT, <italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Body temperature</td><td align="left" valign="top">0.855 (0.815-0.893)</td><td align="left" valign="top">0.671 (0.624-0.713)</td><td align="left" valign="top">0.864 (0.831-0.896)</td><td align="left" valign="top">0.899 (0.868-0.928)</td><td align="left" valign="top">0.863 (0.828-0.897)</td><td align="left" valign="top">0.944 (0.914-0.968)</td></tr><tr><td align="left" valign="top">Fatigue and muscle or joint pain</td><td align="left" valign="top">0.926 (0.846-0.984)</td><td align="left" valign="top">0.744 (0.564-0.884)</td><td align="left" valign="top">0.863 (0.735-0.958)</td><td align="left" valign="top">0.945 (0.880-1.000)</td><td align="left" valign="top">0.824 (0.682-0.931)</td><td align="left" valign="top">0.926 (0.846-0.984)</td></tr><tr><td align="left" valign="top">Headache and dizziness</td><td align="left" valign="top">0.879 (0.784-0.955)</td><td align="left" valign="top">0.757 (0.64-0.853)</td><td align="left" valign="top">0.93 (0.862-0.985)</td><td align="left" valign="top">0.93 (0.862-0.985)</td><td align="left" valign="top">0.93 (0.862-0.985)</td><td align="left" valign="top">0.93 (0.862-0.985)</td></tr><tr><td align="left" valign="top">Respiratory symptoms</td><td align="left" valign="top">0.892 (0.849-0.928)</td><td align="left" valign="top">0.672 (0.607-0.73)</td><td align="left" valign="top">0.925 (0.888-0.955)</td><td align="left" valign="top">0.936 (0.9-0.966)</td><td align="left" valign="top">0.868 (0.825-0.903)</td><td align="left" valign="top">0.938 (0.904-0.967)</td></tr><tr><td align="left" valign="top">Throat appearance</td><td align="left" valign="top">0.731 (0.556-0.861)</td><td align="left" valign="top">0.667 (0.489-0.800)</td><td align="left" valign="top">0.862 (0.750-0.944)</td><td align="left" valign="top">0.824 (0.706-0.914)</td><td align="left" valign="top">0.915 (0.833-0.979)</td><td align="left" valign="top">0.915 (0.824-0.980)</td></tr><tr><td align="left" valign="top">Gastrointestinal symptoms</td><td align="left" valign="top">0.746 (0.588-0.857)</td><td align="left" valign="top">0.85 (0.722-0.950)</td><td align="left" valign="top">0.955 (0.857-1.000)</td><td align="left" valign="top">0.894 (0.780-0.982)</td><td align="left" valign="top">0.837 (0.684-0.957)</td><td align="left" valign="top">0.955 (0.878-1.000)</td></tr><tr><td align="left" valign="top">Other symptoms</td><td align="left" valign="top">0.533 (0.213-0.815)</td><td align="left" valign="top">0 (0-0)</td><td align="left" valign="top">0.750 (0-1.000)</td><td align="left" valign="top">0.800 (0-1.000)</td><td align="left" valign="top">0.500 (0-0.800)</td><td align="left" valign="top">0.750 (0-1.000)</td></tr><tr><td align="left" valign="top">Vaccination information</td><td align="left" valign="top">0.437 (0.324-0.528)</td><td align="left" valign="top">0.327 (0.208-0.437)</td><td align="left" valign="top">0.699 (0.582-0.793)</td><td align="left" valign="top">0.755 (0.657-0.837)</td><td align="left" valign="top">0.776 (0.674-0.863)</td><td align="left" valign="top">0.836 (0.750-0.908)</td></tr><tr><td align="left" valign="top">Onset date</td><td align="left" valign="top">0.168 (0.111-0.22)</td><td align="left" valign="top">0.827 (0.78-0.868)</td><td align="left" valign="top">0.923 (0.891-0.952)</td><td align="left" valign="top">0.895 (0.857-0.929)</td><td align="left" valign="top">0.887 (0.847-0.923)</td><td align="left" valign="top">0.957 (0.934-0.978)</td></tr><tr><td align="left" valign="top">Overall (macroaverage)</td><td align="left" valign="top">0.685 (0.630-0.736)</td><td align="left" valign="top">0.613 (0.570-0.653)</td><td align="left" valign="top">0.863 (0.797-0.902)</td><td align="left" valign="top">0.875 (0.800-0.913)</td><td align="left" valign="top">0.822 (0.752-0.874)</td><td align="left" valign="top">0.906 (0.833-0.945)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Data are from the held-out evaluation set (n=151).</p></fn><fn id="table2fn2"><p><sup>b</sup>FSL: few-shot learning.</p></fn><fn id="table2fn3"><p><sup>c</sup>SFT: supervised fine-tuning.</p></fn></table-wrap-foot></table-wrap><p>In the fine-tuning experiments, both the Google Gemma 2 2B and 27B models showed substantial performance improvements. The 2B model achieved a macroaveraged <italic>F</italic><sub>1</sub>-score of 0.822 (95% CI 0.752-0.874), while the 27B model fine-tuned with QLoRA reached the highest point estimate (0.906, 95% CI 0.833-0.945; specificity of 0.969 and PPV of 0.906). The fine-tuned 27B model&#x2019;s advantage over Claude 3.5 Sonnet was small, and its CI included zero (&#x0394;<italic>F</italic><sub>1</sub>-score=0.030, 95% CI &#x2212;0.035 to 0.140); therefore, the 2 models were not statistically comparable. The 27B model did, however, significantly outperform the smaller fine-tuned 2B model (&#x0394;<italic>F</italic><sub>1</sub>-score=0.084, 95% CI 0.040 to 0.163). Per-category sensitivity, specificity, PPV, and NPV for all models are provided in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s3-2"><title>Performance by Primary Diagnosis</title><p>Because the cohort was dominated by COVID-19 (approximately 80% of notes mentioned COVID-19), a fine-grained diagnosis breakdown was not feasible; we therefore compared COVID-19 (n=120, 79%) and non&#x2013;COVID-19 (n=31, 21%) notes in the test set. The best fine-tuned model was stable across strata (macroaveraged <italic>F</italic><sub>1</sub>-score 0.900 vs 0.893), whereas the rule-based algorithm (0.702 vs 0.605) and Claude 3.5 Sonnet (0.886 vs 0.735) showed lower performance on non&#x2013;COVID-19 notes. Given the small non&#x2013;COVID-19 subgroup, these results are exploratory (Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s3-3"><title>Error Analysis</title><p>A qualitative error analysis of misclassifications by the LLM revealed that many errors occurred when multiple extraction targets were listed consecutively, separated by commas or periods, followed by a collective negation (eg, &#x201C;Diarrhea, nausea negative.&#x201D; and &#x201C;Abdominal pain and vomiting. Wheezing negative.&#x201D;). These cases can be interpreted in multiple ways and are examples of situations in which even human experts would find it difficult to reach a consensus.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>We demonstrated that LLMs can extract infectious disease&#x2013;related information and vaccination data from free-text clinical notes written in Japanese in EHRs with high accuracy. For items with high descriptive variability, LLMs markedly outperformed rule-based regular expressions. The commercial model Claude 3.5 Sonnet (with FSL) reached a macroaveraged <italic>F</italic><sub>1</sub>-score of 0.875 (95% CI 0.800-0.913) without additional training, while fine-tuning the open-source Gemma 2 27B model with QLoRA on only 500 notes achieved a comparable point estimate (0.906, 95% CI 0.833-0.945); the difference between the 2 was small and not statistically distinguishable (&#x0394;<italic>F</italic><sub>1</sub>-score=0.030, 95% CI &#x2212;0.035 to 0.140). The fine-tuned 27B model did, however, significantly outperform the smaller fine-tuned 2B model (&#x0394;<italic>F</italic><sub>1</sub>-score=0.084, 95% CI 0.040 to 0.163). Given the security risks of sending clinical text containing sensitive personal information to external commercial models, achieving comparable performance with a smaller open-source model demonstrates the feasibility of processing data securely within a closed environment, enhancing real-world applicability.</p></sec><sec id="s4-2"><title>Comparison With Prior Work and Interpretation of Findings</title><p>Our findings are consistent with prior reports that LLMs outperform traditional NLP approaches for structured extraction from EHR free text [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref25">25</xref>] and extend them to Japanese primary care narratives, which, to our knowledge, have not previously been targeted directly; prior Japanese clinical NLP has largely used BERT (Bidirectional Encoder Representations from Transformers)-based models or symptom taggers on other document types [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. In contrast to BERT-based COVID-19 case detection in Dutch general practice [<xref ref-type="bibr" rid="ref10">10</xref>] and rule or curation pipelines in large US EHR systems [<xref ref-type="bibr" rid="ref11">11</xref>], we show that a compact, locally deployable, fine-tuned generative model can perform comparably to a commercial model, consistent with recent reports that fine-tuned open-source LLMs reach human-level extraction performance from few examples [<xref ref-type="bibr" rid="ref28">28</xref>] and that local open-source deployment mitigates the privacy risks of sending notes to external services [<xref ref-type="bibr" rid="ref29">29</xref>], while benchmark studies find that fine-tuning still tends to outperform zero or few-shot prompting for extraction tasks [<xref ref-type="bibr" rid="ref30">30</xref>].</p><p>Several categories scored low under FSL (eg, other symptoms and vaccination information for Gemma 2 2B). This likely reflects (1) class imbalance&#x2014;rare symptoms such as chills, dizziness, and rash had very few positive cases; (2) ambiguity in clinical expressions, including consecutively listed findings followed by a collective negation&#x2014;a well-known challenge for negation or assertion detection in clinical text [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]; and (3) the limited capacity of small models without parameter updates. Fine-tuning substantially mitigated these issues. The low interrater agreement observed for some common items (eg, fever), which is consistent with prior reports that annotation of clinical narratives is intrinsically variable [<xref ref-type="bibr" rid="ref33">33</xref>], further indicates that part of the residual error stems from intrinsic annotation ambiguity rather than model failure.</p><p>At this feasibility stage, the extraction is fully automated: the structured outputs are intended for subsequent review and aggregation by public health or clinical staff rather than for autonomous clinical decision-making, and no specialized user interaction with the model is required beyond the domain expertise needed to interpret aggregated surveillance signals.</p></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations. First, the 5 clinics were a convenience sample, and the annotated (analyzed) set overrepresented some clinics while leaving others almost entirely unannotated (eg, 2 clinics contributed almost no analyzable notes); the algorithm&#x2019;s accuracy may not generalize to clinics with different documentation styles. Second, because manual annotation was resource-limited, only approximately 40% of the candidate notes (802 of 2020) were annotated, and these differed systematically from the unannotated notes: annotated patients were older (median age 15 vs 3 years; <italic>P</italic>&#x003C;.001) and came disproportionately from particular clinics (<italic>P</italic>&#x003C;.001), although sex did not differ (<italic>P</italic>=.44). Because the choice of which notes to annotate was not random, this raises the possibility of selection bias. Third, the evaluation set was predominantly single annotated (106/151, 70% of notes), so interrater reliability was quantified on the 216-note pilot subset; a sensitivity analysis showed comparable extraction performance between triple- and single-annotated notes (Table S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Fourth, we could not establish an external validation set, so the results represent internal validation only. Finally, most patients had a single note, so the ability to capture temporal symptom changes is unverified. Critically, this study demonstrates extraction accuracy, not downstream surveillance utility: we did not evaluate whether the extracted signals improve outbreak detection, trend monitoring, or timeliness in real-world settings.</p><p>Future research should (1) validate the algorithm on external clinics not involved in training, (2) link extracted symptom and vaccination signals to observed infectious disease trends to assess surveillance timeliness and accuracy, and (3) extend annotation to longitudinal records.</p></sec><sec id="s4-4"><title>Conclusions</title><p>By fine-tuning the open-source LLM Google Gemma 2, we developed and internally validated a model specialized in extracting and structuring infectious disease&#x2013;related information from Japanese free-text clinical notes with accuracy comparable to that of a commercial model, while remaining deployable in a closed environment. Because the model can run locally without transmitting sensitive clinical text to external services, it offers a practical and privacy-preserving foundation for EHR-based digital surveillance in settings where data cannot leave the institution. Realizing that potential will require external validation and, above all, demonstrating that the extracted signals improve real-world outbreak detection and timeliness&#x2014;the essential next step toward broader public health impact.</p></sec></sec></body><back><ack><p>The authors are grateful to M3 Inc for their dedicated efforts in maintaining and developing the Japan Medical Data Survey database, which was used for this analysis.</p><p>Generative AI (Gemini 2.5 Flash and Pro [Alphabet Inc] and Claude Opus 4.8 [Anthropic PBC]) was used solely for language and grammar polishing of the manuscript. It was not used to generate study data, analyses, results, or scientific content. All authors reviewed and take responsibility for the final text.</p></ack><notes><sec><title>Funding</title><p>This research received no specific grant from any funding agency in the public, commercial, or not-for-profit sectors.</p></sec><sec><title>Data Availability</title><p>The data used in this study were obtained under a restricted-use agreement with M3 Inc and are therefore not freely available to the public. Any inquiries regarding the data can be directed to the corresponding author.</p></sec></notes><fn-group><fn fn-type="con"><p>All authors contributed to the conceptualization of the study and the final review of the manuscript. HY was responsible for the methodology, formal analysis, visualization, and drafting of the original manuscript. In collaboration with HY, YN and YY also performed key work on the methodology and formal analysis. Data curation was carried out by DS, KK, and A Iwata. Validation of the work was performed by HY, HM, YH, NVdV, and A Igarashi. Finally, A Igarashi provided overall supervision for the project. All authors approved the final manuscript and are accountable for all aspects of the work.</p></fn><fn fn-type="conflict"><p>HY, A Igarashi, DS, KK, and A Iwata receive research grants from Moderna Inc, unrelated to this study. NVdV is an employee of Moderna Inc. All other authors declare no other conflicts of interest.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb2">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb3">FSL</term><def><p>few-shot learning</p></def></def-item><def-item><term id="abb4">JAMDAS</term><def><p>Japan Medical Data Survey</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">LoRA</term><def><p>low-rank adaptation</p></def></def-item><def-item><term id="abb7">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb8">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb9">PPV</term><def><p>positive predictive value</p></def></def-item><def-item><term id="abb10">QLoRA</term><def><p>quantized low-rank adaptation</p></def></def-item><def-item><term id="abb11">RSV</term><def><p>respiratory syncytial virus</p></def></def-item><def-item><term id="abb12">SOAP</term><def><p>subjective, objective, assessment, and plan</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boehm</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Wolfe</surname><given-names>MK</given-names> </name><name name-style="western"><surname>White</surname><given-names>BJ</given-names> </name><name name-style="western"><surname>Hughes</surname><given-names>B</given-names> </name><name name-style="western"><surname>Duong</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bidwell</surname><given-names>A</given-names> </name></person-group><article-title>More than a tripledemic: influenza A virus, respiratory syncytial virus, SARS-CoV-2, and human metapneumovirus in wastewater during winter 2022-2023</article-title><source>Environ Sci Technol Lett</source><year>2023</year><volume>10</volume><issue>8</issue><fpage>622</fpage><lpage>627</lpage><pub-id pub-id-type="doi">10.1021/acs.estlett.3c00385</pub-id><pub-id pub-id-type="medline">37577361</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hanage</surname><given-names>WP</given-names> </name><name name-style="western"><surname>Schaffner</surname><given-names>W</given-names> </name></person-group><article-title>Burden of acute respiratory infections caused by influenza virus, respiratory syncytial virus, and SARS-CoV-2 with consideration of older adults: a narrative review</article-title><source>Infect Dis Ther</source><year>2025</year><month>01</month><volume>14</volume><issue>Suppl 1</issue><fpage>5</fpage><lpage>37</lpage><pub-id pub-id-type="doi">10.1007/s40121-024-01080-4</pub-id><pub-id pub-id-type="medline">39739200</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cong</surname><given-names>B</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name></person-group><article-title>The role of respiratory co-infection with influenza or respiratory syncytial virus in the clinical severity of COVID-19 patients: a systematic review and meta-analysis</article-title><source>J Glob Health</source><year>2022</year><month>09</month><day>17</day><volume>12</volume><fpage>05040</fpage><pub-id pub-id-type="doi">10.7189/jogh.12.05040</pub-id><pub-id pub-id-type="medline">36112521</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sugawara</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ohkusa</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ibuka</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kawanohara</surname><given-names>H</given-names> </name><name name-style="western"><surname>Taniguchi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Okabe</surname><given-names>N</given-names> </name></person-group><article-title>Real-time prescription surveillance and its application to monitoring seasonal influenza activity in Japan</article-title><source>J Med Internet Res</source><year>2012</year><month>01</month><day>16</day><volume>14</volume><issue>1</issue><fpage>e14</fpage><pub-id pub-id-type="doi">10.2196/jmir.1881</pub-id><pub-id pub-id-type="medline">22249906</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rhee</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dantes</surname><given-names>R</given-names> </name><name name-style="western"><surname>Epstein</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Incidence and trends of sepsis in US hospitals using clinical vs claims data, 2009-2014</article-title><source>JAMA</source><year>2017</year><month>10</month><day>3</day><volume>318</volume><issue>13</issue><fpage>1241</fpage><lpage>1249</lpage><pub-id pub-id-type="doi">10.1001/jama.2017.13836</pub-id><pub-id pub-id-type="medline">28903154</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jerfy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Selden</surname><given-names>O</given-names> </name><name name-style="western"><surname>Balkrishnan</surname><given-names>R</given-names> </name></person-group><article-title>The growing impact of natural language processing in healthcare and public health</article-title><source>Inquiry</source><year>2024</year><volume>61</volume><fpage>469580241290095</fpage><pub-id pub-id-type="doi">10.1177/00469580241290095</pub-id><pub-id pub-id-type="medline">39396164</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hossain</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rana</surname><given-names>R</given-names> </name><name name-style="western"><surname>Higgins</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Natural language processing in electronic health records in relation to healthcare decision-making: a systematic review</article-title><source>Comput Biol Med</source><year>2023</year><month>03</month><volume>155</volume><fpage>106649</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106649</pub-id><pub-id pub-id-type="medline">36805219</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ntinopoulos</surname><given-names>V</given-names> </name><name name-style="western"><surname>Rodriguez Cetina Biefer</surname><given-names>H</given-names> </name><name name-style="western"><surname>Tudorache</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Large language models for data extraction from unstructured and semi-structured electronic health records: a multiple model performance evaluation</article-title><source>BMJ Health Care Inform</source><year>2025</year><month>01</month><day>19</day><volume>32</volume><issue>1</issue><fpage>e101139</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2024-101139</pub-id><pub-id pub-id-type="medline">39832824</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Rong</surname><given-names>R</given-names> </name><etal/></person-group><article-title>A critical assessment of using ChatGPT for extracting structured data from clinical notes</article-title><source>NPJ Digit Med</source><year>2024</year><month>05</month><day>1</day><volume>7</volume><issue>1</issue><fpage>106</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01079-8</pub-id><pub-id pub-id-type="medline">38693429</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Homburg</surname><given-names>M</given-names> </name><name name-style="western"><surname>Meijer</surname><given-names>E</given-names> </name><name name-style="western"><surname>Berends</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A natural language processing model for COVID-19 detection based on Dutch general practice electronic health records by using bidirectional encoder representations from transformers: development and validation study</article-title><source>J Med Internet Res</source><year>2023</year><month>10</month><day>4</day><volume>25</volume><fpage>e49944</fpage><pub-id pub-id-type="doi">10.2196/49944</pub-id><pub-id pub-id-type="medline">37792444</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wagner</surname><given-names>T</given-names> </name><name name-style="western"><surname>Shweta</surname><given-names>F</given-names> </name><name name-style="western"><surname>Murugadoss</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Augmented curation of clinical notes from a massive EHR system reveals symptoms of impending COVID-19 diagnosis</article-title><source>Elife</source><year>2020</year><month>07</month><day>7</day><volume>9</volume><fpage>e58227</fpage><pub-id pub-id-type="doi">10.7554/eLife.58227</pub-id><pub-id pub-id-type="medline">32633720</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferraro</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Ye</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gesteland</surname><given-names>PH</given-names> </name><etal/></person-group><article-title>The effects of natural language processing on cross-institutional portability of influenza case detection for disease surveillance</article-title><source>Appl Clin Inform</source><year>2017</year><month>05</month><day>31</day><volume>8</volume><issue>2</issue><fpage>560</fpage><lpage>580</lpage><pub-id pub-id-type="doi">10.4338/ACI-2016-12-RA-0211</pub-id><pub-id pub-id-type="medline">28561130</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Levin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Finley</surname><given-names>PD</given-names> </name><name name-style="western"><surname>Heilig</surname><given-names>CM</given-names> </name></person-group><article-title>Chief complaint classification with recurrent neural networks</article-title><source>J Biomed Inform</source><year>2019</year><month>05</month><volume>93</volume><fpage>103158</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2019.103158</pub-id><pub-id pub-id-type="medline">30926471</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hardjojo</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gunachandran</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Validation of a natural language processing algorithm for detecting infectious disease symptoms in primary care electronic medical records in Singapore</article-title><source>JMIR Med Inform</source><year>2018</year><month>06</month><day>11</day><volume>6</volume><issue>2</issue><fpage>e36</fpage><pub-id pub-id-type="doi">10.2196/medinform.8204</pub-id><pub-id pub-id-type="medline">29907560</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>MacRae</surname><given-names>J</given-names> </name><name name-style="western"><surname>Love</surname><given-names>T</given-names> </name><name name-style="western"><surname>Baker</surname><given-names>MG</given-names> </name><etal/></person-group><article-title>Identifying influenza-like illness presentation from unstructured general practice clinical narrative using a text classifier rule-based expert system versus a clinical expert</article-title><source>BMC Med Inform Decis Mak</source><year>2015</year><month>10</month><day>6</day><volume>15</volume><fpage>78</fpage><pub-id pub-id-type="doi">10.1186/s12911-015-0201-3</pub-id><pub-id pub-id-type="medline">26445235</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miyawaki</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kitajima</surname><given-names>K</given-names> </name><name name-style="western"><surname>Iwata</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sato</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tsugawa</surname><given-names>Y</given-names> </name></person-group><article-title>Antibiotic prescription for outpatients with COVID-19 in primary care settings in Japan</article-title><source>JAMA Netw Open</source><year>2023</year><month>07</month><day>3</day><volume>6</volume><issue>7</issue><fpage>e2325212</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.25212</pub-id><pub-id pub-id-type="medline">37490294</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>Corporate information</article-title><source>M3, Inc</source><access-date>2025-06-25</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://corporate.m3.com/en/corporate/">https://corporate.m3.com/en/corporate/</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="web"><article-title>Claude 3.5 Sonnet</article-title><source>Anthropic</source><year>2024</year><access-date>2025-06-25</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.anthropic.com/news/claude-3-5-sonnet">https://www.anthropic.com/news/claude-3-5-sonnet</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><collab>Gemma Team</collab><name name-style="western"><surname>Riviere</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pathak</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sessa</surname><given-names>PG</given-names> </name><etal/></person-group><article-title>Gemma 2: improving open language models at a practical size</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 31, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.00118</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Mann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ryder</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Language models are few-shot learners</article-title><source>arXiv</source><comment>Preprint posted online on  May 28, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wallis</surname><given-names>P</given-names> </name><etal/></person-group><article-title>LoRA: low-rank adaptation of large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 17, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2106.09685</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Dettmers</surname><given-names>T</given-names> </name><name name-style="western"><surname>Pagnoni</surname><given-names>A</given-names> </name><name name-style="western"><surname>Holtzman</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zettlemoyer</surname><given-names>L</given-names> </name></person-group><article-title>QLoRA: efficient finetuning of quantized LLMs</article-title><source>arXiv</source><comment>Preprint posted online on  May 23, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2305.14314</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Wolf</surname><given-names>T</given-names> </name><name name-style="western"><surname>Debut</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sanh</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Transformers: state-of-the-art natural language processing</article-title><source>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations</source><year>2020</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>38</fpage><lpage>45</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.emnlp-demos.6</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>von Werra</surname><given-names>L</given-names> </name><name name-style="western"><surname>Belkada</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tunstall</surname><given-names>L</given-names> </name><etal/></person-group><article-title>TRL: transformer reinforcement learning</article-title><source>GitHub</source><year>2020</year><access-date>2026-08-21</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/huggingface/trl">https://github.com/huggingface/trl</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Du</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Improving large language models for clinical named entity recognition via prompt engineering</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>09</month><day>1</day><volume>31</volume><issue>9</issue><fpage>1812</fpage><lpage>1820</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad259</pub-id><pub-id pub-id-type="medline">38281112</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kawazoe</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shibata</surname><given-names>D</given-names> </name><name name-style="western"><surname>Shinohara</surname><given-names>E</given-names> </name><name name-style="western"><surname>Aramaki</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ohe</surname><given-names>K</given-names> </name></person-group><article-title>A clinical specific BERT developed using a huge Japanese clinical text corpus</article-title><source>PLoS One</source><year>2021</year><volume>16</volume><issue>11</issue><fpage>e0259763</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0259763</pub-id><pub-id pub-id-type="medline">34752490</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nishiyama</surname><given-names>T</given-names> </name><name name-style="western"><surname>Yamaguchi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Han</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Automated system to capture patient symptoms from multitype Japanese clinical texts: retrospective study</article-title><source>JMIR Med Inform</source><year>2024</year><month>09</month><day>24</day><volume>12</volume><fpage>e58977</fpage><pub-id pub-id-type="doi">10.2196/58977</pub-id><pub-id pub-id-type="medline">39316418</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lian</surname><given-names>L</given-names> </name><name name-style="western"><surname>Hao</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Human level information extraction from clinical reports with finetuned language models</article-title><source>Sci Rep</source><year>2025</year><volume>15</volume><issue>1</issue><fpage>45239</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-28767-z</pub-id><pub-id pub-id-type="medline">41286063</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Builtjes</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bosma</surname><given-names>J</given-names> </name><name name-style="western"><surname>Prokop</surname><given-names>M</given-names> </name><name name-style="western"><surname>van Ginneken</surname><given-names>B</given-names> </name><name name-style="western"><surname>Hering</surname><given-names>A</given-names> </name></person-group><article-title>Leveraging open-source large language models for clinical information extraction in resource-constrained settings</article-title><source>JAMIA Open</source><year>2025</year><month>10</month><volume>8</volume><issue>5</issue><fpage>ooaf109</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf109</pub-id><pub-id pub-id-type="medline">41041625</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Benchmarking large language models for biomedical natural language processing applications and recommendations</article-title><source>arXiv</source><comment>Preprint posted online on  May 10, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2305.16326</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chapman</surname><given-names>WW</given-names> </name><name name-style="western"><surname>Bridewell</surname><given-names>W</given-names> </name><name name-style="western"><surname>Hanbury</surname><given-names>P</given-names> </name><name name-style="western"><surname>Cooper</surname><given-names>GF</given-names> </name><name name-style="western"><surname>Buchanan</surname><given-names>BG</given-names> </name></person-group><article-title>A simple algorithm for identifying negated findings and diseases in discharge summaries</article-title><source>J Biomed Inform</source><year>2001</year><month>10</month><volume>34</volume><issue>5</issue><fpage>301</fpage><lpage>310</lpage><pub-id pub-id-type="doi">10.1006/jbin.2001.1029</pub-id><pub-id pub-id-type="medline">12123149</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harkema</surname><given-names>H</given-names> </name><name name-style="western"><surname>Dowling</surname><given-names>JN</given-names> </name><name name-style="western"><surname>Thornblade</surname><given-names>T</given-names> </name><name name-style="western"><surname>Chapman</surname><given-names>WW</given-names> </name></person-group><article-title>ConText: an algorithm for determining negation, experiencer, and temporal status from clinical reports</article-title><source>J Biomed Inform</source><year>2009</year><month>10</month><volume>42</volume><issue>5</issue><fpage>839</fpage><lpage>851</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2009.05.002</pub-id><pub-id pub-id-type="medline">19435614</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raghavan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Fosler-Lussier</surname><given-names>E</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>AM</given-names> </name></person-group><article-title>Inter-annotator reliability of medical events, coreferences and temporal relations in clinical narratives by annotators with varying levels of clinical expertise</article-title><source>AMIA Annu Symp Proc</source><year>2012</year><volume>2012</volume><fpage>1366</fpage><lpage>1374</lpage><pub-id pub-id-type="medline">23304416</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Hyperparameters, note templates, per-category sensitivity/specificity/positive predictive value/negative predictive value with 95% CIs, interrater agreement, annotated-vs-unannotated and per-clinic characteristics, and performance by diagnosis and by annotation depth.</p><media xlink:href="formative_v10i1e84974_app1.docx" xlink:title="DOCX File, 2385 KB"/></supplementary-material></app-group></back></article>