<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e86647</article-id><article-id pub-id-type="doi">10.2196/86647</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Toward Automating the Selection of Articles Reporting EQ-5D Data for Systematic Literature Reviews Using Large Language Models: Algorithm Development and Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Kert&#x00E9;sz</surname><given-names>G&#x00E1;bor</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Czere</surname><given-names>J&#x00E1;nos Tibor</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zrubka</surname><given-names>Zsombor</given-names></name><degrees>PhD, MD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gulacsi</surname><given-names>Laszlo</given-names></name><degrees>DSC, PhD, MD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pentek</surname><given-names>Marta</given-names></name><degrees>PhD, DSC, MD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff5">5</xref></contrib></contrib-group><aff id="aff1"><institution>Software Engineering Institute, John von Nemann Faculty of Informatics, Obuda University</institution><addr-line>Becsi str 96/B</addr-line><addr-line>Budapest</addr-line><country>Hungary</country></aff><aff id="aff2"><institution>Laboratory of Parallel and Distributed Systems, Institute for Computer Science and Control</institution><addr-line>Budapest</addr-line><country>Hungary</country></aff><aff id="aff3"><institution>Doctoral School of Innovation Management, Obuda University</institution><addr-line>Budapest</addr-line><country>Hungary</country></aff><aff id="aff4"><institution>PSI CRO Hungary LLC</institution><addr-line>Budapest</addr-line><country>Hungary</country></aff><aff id="aff5"><institution>HECON Health Economics Research Center, Obuda University</institution><addr-line>Budapest</addr-line><country>Hungary</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Chehab</surname><given-names>Ali</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to G&#x00E1;bor Kert&#x00E9;sz, PhD, Software Engineering Institute, John von Nemann Faculty of Informatics, Obuda University, Becsi str 96/B, Budapest, 1034, Hungary, +3616665528; <email>kertesz.gabor@nik.uni-obuda.hu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>31</day><month>8</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e86647</elocation-id><history><date date-type="received"><day>19</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>08</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>10</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; G&#x00E1;bor Kert&#x00E9;sz, J&#x00E1;nos Tibor Czere, Zsombor Zrubka, Laszlo Gulacsi, Marta Pentek. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 31.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e86647"/><abstract><sec><title>Background</title><p>Systematic literature reviews (SLRs) are essential for evidence synthesis in health research but remain labor-intensive, especially at the screening stage. Manual review of titles and abstracts requires substantial human effort, while existing automation tools still have limited adoption in health technology assessment. The EQ-5D questionnaire, a widely used patient-reported outcome measure for health-related quality of life, provides data that frequently underpin reimbursement and policy decisions.</p></sec><sec><title>Objective</title><p>This pilot study evaluated whether recent large language models (LLMs) can support the identification of publications reporting EQ-5D data in PubMed records, using only publicly available metadata (title, abstract, and keywords).</p></sec><sec sec-type="methods"><title>Methods</title><p>A total of 200 publications retrieved through the EuroQol PubMed filter were manually labeled by experts as reporting or not reporting EQ-5D data. The dataset was split into stratified training, validation, and test subsets. Several machine learning approaches were compared, including a Na&#x00EF;ve Bayes baseline using bag-of-words features, a decision-tree model based on full-text keyword occurrence, and transformer-based LLMs (Bidirectional Encoder Representations from Transformers [BERT], Biomedical BERT [BioBERT], Scientific BERT [SciBERT], and Biomedical Language Understanding Evaluation BERT [BlueBERT]). Both classifier-only and fine-tuned configurations were tested across multiple learning rates. Model performance was assessed using accuracy, precision, recall, and <italic>F</italic><sub>1</sub>-score.</p></sec><sec sec-type="results"><title>Results</title><p>Baseline approaches achieved near-random test performance (accuracy around 0.53). Classifier-only LLMs modestly improved results (accuracy up to 0.64 with SciBERT). Fine-tuned models substantially outperformed these baselines, with BERT and BioBERT achieving the best performance (accuracy=0.70; <italic>F</italic><sub>1</sub>-score=0.68). In screening-oriented evaluation, this configuration achieved 90.0% sensitivity, 40.0% specificity, and 6 false negatives on the held-out test set. The models reproduced human screening tendencies despite the small dataset size, demonstrating the technical feasibility of LLM-assisted article selection.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This study provides the first demonstration of LLM-assisted identification of EQ-5D data in biomedical literature. The findings support technical feasibility but do not establish a reliable stand-alone automated screening tool. Although limited by dataset size, the proposed workflow is reproducible and adaptable to other patient-reported outcome measures. Because validation was based on a single small train-validation-test split, the results should be interpreted as preliminary; future work will scale data collection, include statistical testing, and explore semisupervised learning to further reduce manual screening workload.</p></sec></abstract><kwd-group><kwd>BERT</kwd><kwd>BioBERT</kwd><kwd>large language model</kwd><kwd>systematic literature review</kwd><kwd>PubMed</kwd><kwd>health informatics</kwd><kwd>EQ-5D</kwd><kwd>quality of life</kwd><kwd>Bidirectional Encoder Representations from Transformers</kwd><kwd>Biomedical Bidirectional Encoder Representations from Transformers</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Systematic literature reviews (SLRs) are scientific methods to collate the available evidence, in order to answer specific research questions and support decisions. SLRs were first used in medicine to inform medical decision-making but currently are widely applied in various fields.</p><p>The methodology for conducting SLRs using electronic literature databases has been standardized to ensure reliable, high-quality, and reproducible findings and thus conclusions [<xref ref-type="bibr" rid="ref1">1</xref>]. However, implementing the SLR method can be resource-intensive in terms of both time and qualified human effort [<xref ref-type="bibr" rid="ref2">2</xref>]. Tasks such as constructing the search strategy, selecting relevant studies through multiple rounds with at least 2 independent reviewers, and extracting data require considerable resources. While the search is performed electronically using the search strategy designed for a specific research question and considering the features of the electronic literature database, the selection of records is a manual process. Two experts review the hits of the search independently from each other and assess which records fulfill the predefined eligibility criteria. This assessment is done first by the publication&#x2019;s title, abstract, and keywords. For publications for which the decision regarding their eligibility was not possible in this round, undergo full-text review. In both phases discrepancies between the 2 independent reviewers are solved by discussions and a third reviewer can also be involved. This sophisticated selection method ensures, as far as possible, avoiding selection errors. However, the process becomes particularly challenging when the search yields a large number of results, such as several thousand.</p><p>These challenges have prompted the development of automation tools based on machine learning (ML) methods in the field of SLRs. The adoption of automation is still limited [<xref ref-type="bibr" rid="ref3">3</xref>]; however, the literature on SLR automation is extensive and continually evolving [<xref ref-type="bibr" rid="ref4">4</xref>]. Text classification methods have [<xref ref-type="bibr" rid="ref5">5</xref>] improved since the breakthroughs of deep learning [<xref ref-type="bibr" rid="ref6">6</xref>] and especially since recent advances with large language models (LLMs) [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Classifying text differs from general classification tasks, as text&#x2014;represented as a sequence of characters&#x2014;cannot be simply converted to a set of features. Early or classical approaches were based on character or word frequency, ignoring word order and sentences themselves. To improve performance on natural language understanding, researchers created models that represent semantic relationships in word sequences [<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Initially, statistical tools formed the foundation of text analysis, with the bag-of-words (BoW) model being an example. This technique serves as a fundamental building block for various other tools. In this approach, the frequency of words in the text is counted, providing a representation based on their occurrence. While this method does not consider the specific positions of words in the text, it can be used as a feature extraction technique. Also, comparing word frequencies can indicate similarity between 2 texts; if the words and their frequencies match, the texts are considered similar [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>A more effective solution involves representing words with their sequential order. Word embedding, or word vectorization, is an approach that represents documents and words as numeric vectors. This representation allows words with similar meanings to have similar vector representations and enables approximation of word meaning in a lower-dimensional space [<xref ref-type="bibr" rid="ref11">11</xref>]. The input to this approach includes the word itself and its position within the text, which can be seen as a time series representation where words must follow each other in the same order.</p><p>However, it is worth mentioning that highly effective general models have only become available with the introduction of deep learning. Multiple methods for natural language processing (NLP) were developed, such as eLMo [<xref ref-type="bibr" rid="ref12">12</xref>], Bidirectional Encoder Representations from Transformers (BERT) [<xref ref-type="bibr" rid="ref13">13</xref>], and GPT [<xref ref-type="bibr" rid="ref14">14</xref>]. eLMo, an NLP framework developed by AllenNLP, uses a 2-layer bidirectional language model (biLM) to calculate word vectors. The biLM consists of both forward and backward passes in each layer. eLMo generates embeddings for a word by considering the entire sentence in which the word appears. GPTs are a framework in the field of generative AI and fall under the category of LLMs. These models are artificial neural networks that use the transformer [<xref ref-type="bibr" rid="ref15">15</xref>] architecture. Applications of GPT are well-known for their ability to generate human-like content.</p><p>BERT is a family of language models, introduced in 2018 by Google researchers. BERT is an unsupervised language representation model that deeply incorporates bidirectionality. It is pretrained on pure plain text corpora, considering the context surrounding each occurrence of a word. As a result, BERT generates contextualized embeddings that vary depending on the sentence. BERT is considered a more suitable approach for text classification compared to GPT, as the bidirectional processing of text allows the method to capture the context deeply; the latter is mostly used for generation.</p><p>The focus of our research is on the automation of SLRs in the field of health, specifically on a health outcome measure, the EQ-5D. Assessing health improvements from the perspective of patients has become a fundamental aspect of health care and the development of health technologies. Patient-reported outcome measures (PROMs) offer valuable insights into changes in health, functional status, and health-related quality of life (HRQoL) as perceived by patients [<xref ref-type="bibr" rid="ref16">16</xref>]. PROMs are typically standardized, self-reported validated questionnaires that enable reliable and valid assessments of health outcomes.</p><p>The EQ-5D questionnaire is a PROM designed to evaluate and measure health. It was initially developed by the EuroQol Group in 1990 [<xref ref-type="bibr" rid="ref17">17</xref>], and since then, numerous studies have used this measurement tool and its subsequent versions [<xref ref-type="bibr" rid="ref18">18</xref>]. The advantage of EQ-5D among PROMs lies in its generic nature, allowing assessment and comparison of health status among patient populations with diverse diseases, as well as the general population. Additionally, EQ-5D data are commonly used to calculate health gains, expressed as quality-adjusted life years (QALYs), in health economic evaluations, thereby informing reimbursement and health policy decisions [<xref ref-type="bibr" rid="ref19">19</xref>]. Consequently, there is a growing demand for SLRs that specifically focus on EQ-5D studies across various domains, with the aim of guiding clinical decision-making, developing public health strategies, conducting health economic evaluations, and performing health technology assessments. Despite the outstanding importance of access to EQ-5D data, to the best of our knowledge, no SLR automation tool specifically focusing on EQ-5D studies has been developed yet.</p><p>The aim of this pilot study was to evaluate the feasibility of automating the selection of publications reporting EQ-5D data based on publicly available metadata (title, abstract, and keywords) using LLMs. Rather than proposing a novel algorithm, the study focuses on the applicability of recent language model architectures to a domain-specific screening task in systematic reviews. Lessons learned from this small-scale experiment are intended to guide future methodological development and larger-scale validation studies.</p><p>A large electronic database of biomedical literature, PubMed [<xref ref-type="bibr" rid="ref20">20</xref>], was searched for EQ-5D studies. The curated and annotated dataset based on human selection work will be used as input for supervised learning, specifically a binary classification task, aimed at predicting whether EQ-5D data are reported in the full text of a given study by analyzing its title, abstract, and keywords. Different LLMs and training methods will be evaluated. Finally, lessons learned from this small experiment are discussed, and some points to consider in future research are formulated. An overview of the methods applied in the study is briefly summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>A brief overview of the methods applied during the research. BoW-NB: bag-of-words&#x2013;Naive Bayes; DT: decision tree; LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e86647_fig01.png"/></fig></sec><sec id="s1-2"><title>Related Work</title><sec id="s1-2-1"><title>Transformer-Based Text Classification</title><p>Recent methods for text-based classification tasks are usually based on transformers [<xref ref-type="bibr" rid="ref15">15</xref>], as these methods seem to outperform any other approach on widely used benchmark datasets [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. For text classification, BERT is preferred over GPT architectures, as GPT models are autoregressive, while BERT is trained bidirectionally [<xref ref-type="bibr" rid="ref25">25</xref>].</p><p>Applications of BERT for text classification tasks are used in many domains: from hate-speech detection in social media [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>] to categorizing medical reports [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. To apply a pretrained BERT model for a text classification task, there are a few approaches. Devlin et al [<xref ref-type="bibr" rid="ref13">13</xref>] presented in the original paper describing BERT that fine-tuning is a straightforward method to specialize the pretrained model on a given task; source codes demonstrating these capabilities were also published. Lee and Hsiang [<xref ref-type="bibr" rid="ref30">30</xref>] applied fine-tuning for patent classification, creating PatentBERT, which outperformed the state-of-the-art DeepPatent. Sun et al [<xref ref-type="bibr" rid="ref31">31</xref>] presented that fine-tuning should be performed on a low learning rate to avoid catastrophic forgetting phenomena. Adhikari et al [<xref ref-type="bibr" rid="ref32">32</xref>] applied the method to classify documents, outperforming other approaches on the benchmark datasets. Zheng and Yang [<xref ref-type="bibr" rid="ref33">33</xref>] proposed a novel BERT-CNN method to improve classification performance.</p><p>Another approach for text classification is based on BERT-based feature extraction; in this case, the parameters in the pretrained model are untouched, and an additional layer or layers are added for classification. BERTScore [<xref ref-type="bibr" rid="ref34">34</xref>] applied BERT to provide contextual embeddings to replace word embeddings, resulting in slightly different word representations based on context. Reimers and Gurevych [<xref ref-type="bibr" rid="ref35">35</xref>] presented Sentence-BERT, which is a fine-tuned BERT model trained in a triplet network architecture for sentence embedding; the triplet loss function is based on the BERT feature vectors for the given anchor, positive, and negative pair.</p></sec><sec id="s1-2-2"><title>Fine-Tuning of Language Models</title><p>While BERT itself is a model pretrained on general data [<xref ref-type="bibr" rid="ref36">36</xref>], there are existing pretrained alternatives with scientific or biomedical domains, such as Scientific BERT (SciBERT) [<xref ref-type="bibr" rid="ref37">37</xref>], Biomedical BERT (BioBERT) [<xref ref-type="bibr" rid="ref38">38</xref>], or Biomedical Language Understanding Evaluation BERT (BlueBERT) [<xref ref-type="bibr" rid="ref39">39</xref>]. It is also worth mentioning the Longformer [<xref ref-type="bibr" rid="ref40">40</xref>], which is a transformer-based architecture overriding the typical token limitation of 512, resulting in high-performing models with 4096 or 16,384 sequence lengths. Other approaches on handling long sequences start with truncating; in many cases, simply truncating the sentence has only minimal effect on performance [<xref ref-type="bibr" rid="ref41">41</xref>].</p><p>Regarding the training of BERT-based models, different methods were applied to improve performance. On fine-tuning, it is recommended to optimize the learning rate hyperparameter, as different settings have a significant impact on the resulting model [<xref ref-type="bibr" rid="ref42">42</xref>]. It is worth mentioning that in multiple studies, the behavior of fine-tuning pretrained models is analyzed [<xref ref-type="bibr" rid="ref43">43</xref>], and addressing instability [<xref ref-type="bibr" rid="ref44">44</xref>] is a key question to achieve peak performance.</p></sec><sec id="s1-2-3"><title>Screening Automation for SLRs</title><p>In the modern era of deep learning, and especially in the era of LLMs, multiple tools have emerged to support study screening for literature reviews. For instance, studies using active learning and classical text representations [<xref ref-type="bibr" rid="ref45">45</xref>-<xref ref-type="bibr" rid="ref47">47</xref>] (eg, Term Frequency-Inverse Document Frequency [TF-IDF] with logistic regression or Na&#x00EF;ve Bayes) reported workload savings at 95% recall (WSS@95) in the range of 60%&#x2010;90%, although performance varies considerably across datasets and domains. GPT-based and related LLMs have recently been explored as tools to automate SLR screening without the need for task-specific training.</p><p>Studies have recently applied ChatGPT (OpenAI) for similar tasks [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>]. A recent study by Guo et al [<xref ref-type="bibr" rid="ref50">50</xref>] presented an automatic study screening method based on ChatGPT, with a problem very similar to that presented in this study. Results showed that the pretrained GPT-4 model as the backbone of the ChatGPT sessions provided high performance. Dennst&#x00E4;dt et al [<xref ref-type="bibr" rid="ref51">51</xref>] used 4 open-source LLMs (<italic>Flan-T5, OpenHermes, Mixtrial,</italic> and <italic>Platypus2</italic>) for title/abstract screening on biomedical reviews. They achieved promising sensitivity (82%&#x2010;98%) but at the cost of lower specificity (eg, 19%&#x2010;75%), indicating many false positives (FPs).</p><p>In parallel, the emergence of LLMs has enabled zero-shot and few-shot screening without task-specific fine-tuning [<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>]. Several recent studies have evaluated models such as GPT-3.5 and GPT-4 in systematic review workflows. Across both BERT-based and LLM-based approaches, a consistent pattern emerges: while transformer models can substantially improve screening efficiency and, in some cases, approximate human-level agreement, their performance is highly context-dependent and often unstable across datasets [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref54">54</xref>].</p><p>Overall, the SLR screening literature [<xref ref-type="bibr" rid="ref55">55</xref>] provides strong evidence that ML can effectively support the process of study selection, particularly through prioritization and human-in-the-loop workflows. However, the task formulation in these studies differs from the presented work; instead of identifying studies that meet general inclusion criteria, they aim to predict reviewer decisions at the document level. In contrast, our setting requires predicting whether a publication contains extractable outcome data based solely on metadata, which represents a classification problem.</p></sec></sec><sec id="s1-3"><title>EQ-5D Prediction</title><p>Regarding screening for EQ-5D, existing methods are designed to retrieve relevant studies [<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref57">57</xref>] (eg, those reporting utility values or economic evaluations). Prediction or classification of individual records based on whether they contain extractable EQ-5D outcome data (eg, index values, visual analog scale (VAS) scores, baseline/follow-up measures, or group differences) instead of mentioning this measure is quite different.</p><p>To the best of our knowledge, there is no well-documented prior work that directly addresses this using automated methods; we did not identify studies that target the classification of publication metadata into those that contain EQ-5D outcomes and those that only reference the instrument.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>EQ-5D Measurement Tool</title><p>The EQ-5D refers to a family of instrument versions of the original EQ-5D measurement tool. The original EQ-5D comprises 2 parts: a descriptive system and a VAS (EQ VAS) [<xref ref-type="bibr" rid="ref58">58</xref>]. The descriptive system covers 5 health domains (mobility, self-care, usual activities, pain/discomfort, and anxiety/depression). The respondent is asked to indicate on a 3-level response scale the problem level that best describes his or her current health status (1: no problem, 2: moderate problem, and 3: extreme problem). Thus, 243 different health states (profiles) can be obtained by completing the descriptive system. Results can be presented as the proportion of the sample indicating different problem levels in each EQ-5D domain. An EQ-5D index score can also be linked to each health profile that reflects the utility (desirability and preference) the society attaches to each health state, where index score 1 represents perfect health, zero refers to the state of death, and negative values indicate health states that are considered as worse than death. The set of EQ-5D index scores (value set and tariffs) is typically country-specific, as it is obtained from the general population in a separate study (using direct utility measurements, mainly the time-trade-off method) [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref59">59</xref>]. The value set once established can be applied directly to calculate the EQ-5D index score from the responses obtained on the descriptive system.</p><p>The second part of the EQ-5D is the EQ VAS, a vertical 20 cm VAS, with end points of 0 and 100, representing the worst and 100 the best health, respectively, the respondent can imagine. Respondents are asked to mark on the EQ VAS how their health is on that day.</p><p>Since the development of the original EQ-5D, new versions have been validated and published, aiming to increase the sensitivity of the tool by applying 5-level response options (EQ-5D-5L version; hence, the original version was renamed as EQ-5D-3L), as well as to measure the HRQoL of children and adolescents with a &#x201C;youth&#x201D; version (EQ-5D-Y-3L and EQ-5D-Y-5L), which uses language adapted for children to describe health problems [<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref61">61</xref>]. All these versions retained the original structure (descriptive system and EQ VAS). In parallel, different modes of administration (eg, digital and phone interviews) have been developed and the versions have been translated into more than 170 languages.</p></sec><sec id="s2-2"><title>Eligibility Criteria</title><p>Studies that report EQ-5D data from patients or the general population are considered eligible. Studies reporting results obtained on the EQ-5D descriptive system (ie, problem levels on the health dimensions) or reporting EQ-5D index scores are eligible, using either version (eg, EQ-5D-3L, EQ-5D-5L, and EQ-5D-Y) and administration method (eg, digital or paper-based or voice interactive system forms, self-complete or proxy version) of the instrument. Conceptual or methodological studies reporting the valuation, development, or psychometric properties of EQ-5D, or those that do not report index scores or descriptive results from actual patients or individuals, are considered out of scope.</p><p>Although EQ VAS is part of the EQ-5D instrument family, studies reporting only EQ VAS values were considered out of scope because the target of this screening task was the identification of studies reporting EQ-5D descriptive-system results or index scores suitable for health-state utility extraction and health economic evidence synthesis.</p><p>No restrictions are applied on the population in which EQ-5D was used; any age group, sex, condition, or geography, etc, is considered. Only full-text studies available in the English language are considered eligible.</p></sec><sec id="s2-3"><title>Data Source</title><p>PubMed [<xref ref-type="bibr" rid="ref20">20</xref>] is a publicly available literature database containing over 36 million citations and abstracts of biomedical, life, and related sciences. PubMed is maintained and regularly updated with new publications and can be searched electronically. PubMed does not include the full text of the journals but generally provides links to relevant sources.</p><p>The EuroQoL Group developed a search filter to identify EQ-5D publications in PubMed. The filter is freely available on the EuroQoL website and can be combined with further search terms (eg, disease-specific terms). The exact search strategy is as follows:</p><disp-formula id="equWL1"><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtext>(</mml:mtext><mml:mtext>euroqol</mml:mtext><mml:mtext>[All Fields] OR eq-5d[All Fields] OR eq5d[All Fields])</mml:mtext></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>We conducted a search in PubMed using the EuroQoL search filter (without adding any further search terms) on October 13, 2022. The search resulted in a total of 15,547 records, published between 1990 and 2022. Altogether 200 publications were randomly selected from the results using the built-in sample command of Stata (version 2017; StataCorp LLC) statistical software. This set of 200 publications (published between 1999 and 2022) is used in the analyses.</p></sec><sec id="s2-4"><title>Data Preprocessing</title><sec id="s2-4-1"><title>Manual Review of the Set of Publications</title><p>All the 200 selected studies were collected in full text and examined by 2 independent reviewers based on the predefined eligibility criteria. Results were matched, and differences were discussed until a concealed position was reached. All publications were labeled indicating whether the publication includes EQ-5D data or not (true/false). All the resulting records consist of these values:</p><list list-type="bullet"><list-item><p>Title</p></list-item><list-item><p>Abstract</p></list-item><list-item><p>Keywords</p></list-item><list-item><p>Label (&#x201C;true&#x201D;/&#x201C;false&#x201D;).</p></list-item></list><p>It is important to note that while all PubMed exported records (metadata) are in English, the full text of the study may be in a different language. According to the predefined eligibility criteria, such studies were labeled as negative by the reviewers. This represents a practical limitation of metadata-based screening, since eligibility may depend on information not fully captured in the metadata.</p></sec><sec id="s2-4-2"><title>Splitting the Data Into Training and Test Datasets</title><p>As the dataset is relatively small considering the varied contents and semantic diversity in scientific texts, 50% of the set is dedicated as the test subset, which was not directly involved in training; it was only used for performance measurement.</p><p>The remaining 100 records were split into a training and a validation subset with the lengths of 85 and 15, respectively. The training subset was used to tune and fine-tune the parameters of the model, while the validation subset was applied indirectly to forecast overfitting. The validation subset contains preselected elements, using a stratified splitting. Stratification was implemented using a fixed validation record ID set, resulting in the same ratio of positive and negative records as in the case of the training set.</p><p>As this is a pilot dataset, with a low number of samples, model performance estimates may be sensitive to the specific splitting strategy. Therefore, our setup is mainly used to provide an initial feasibility report.</p></sec></sec><sec id="s2-5"><title>Baseline Analysis: BoW Model</title><p>Before applying advanced deep learning techniques, it is advised to check the performance of simple approaches as a baseline for our experiments. In the case of text classification, the previously mentioned BoW method is a straightforward selection.</p><p>This technique operates by systematically compiling the words within the text and counting their respective frequencies. Word occurrences define a feature vector for each record, which can be applied as an input for a simple classifier. Let&#x2019;s define <italic>V</italic> vocabulary of words, with <italic>w</italic><sub><italic>j</italic></sub> representing the j<sup><italic>th</italic></sup> word in <italic>V</italic>. Let <italic>x<sub>i</sub></italic> &#x2208; <italic>X</italic> represent the input data, and <italic>t<sub>i</sub></italic> &#x2208; <italic>T</italic> the input texts, where</p><disp-formula id="equWL2"><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mtext>=</mml:mtext><mml:mtext>count</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>for all,</p><p>A Na&#x00EF;ve Bayes classifier based on the BoW methodology is trained and evaluated across identical subsets. The probability is based on the Bayes-theorem, having</p><disp-formula id="equWL4"><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mtext>|</mml:mtext><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mtext>=</mml:mtext><mml:mfrac><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mtext>|</mml:mtext><mml:mi>c</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>c</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn1"><mml:mi>P</mml:mi><mml:mo>(</mml:mo><mml:mi>a</mml:mi><mml:mtext>|</mml:mtext><mml:mi>b</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula> represents the conditional probability of event <italic>a</italic> occurring if event <italic>b</italic> has occurred; and <italic>c</italic> defining the binary class of a negative or a positive sample, <italic>c &#x2208; C</italic>.</p><p>The Na&#x00EF;ve Bayes classifier is a simple yet efficient method, which is based on a few assumptions regarding the data, features are independent and of equal importance. We applied a Gaussian Na&#x00EF;ve Bayes classifier, where <inline-formula><mml:math id="ieqn2"><mml:mi>P</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mtext>|</mml:mtext><mml:mi>c</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula> is assumed to be following a Gaussian distribution. Based on the standard univariate Gaussian distribution,</p><disp-formula id="equWL5"><mml:math id="eqn4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mtext>|</mml:mtext><mml:mi>c</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mtext>=</mml:mtext><mml:mfrac><mml:mn>1</mml:mn><mml:msqrt><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:msubsup><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:msqrt></mml:mfrac><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>-</mml:mtext><mml:mfrac><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mtext>-</mml:mtext><mml:msub><mml:mi>&#x03BC;</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mn>2</mml:mn><mml:msubsup><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <italic>&#x03BC;</italic> represents the mean and <italic>&#x03C3;</italic> represents the variance of the count of word <italic>w</italic><sub><italic>j</italic></sub> in class <italic>c</italic>.</p><p>When classifying, <italic>P</italic> (<italic>x</italic><sub><italic>i</italic></sub>) can be ignored, as we assume that it is constant for all classes, <italic>P</italic> (<italic>c</italic>) can be given as</p><disp-formula id="equWL6"><mml:math id="eqn5"><mml:mi>P</mml:mi><mml:mo>(</mml:mo><mml:mi>c</mml:mi><mml:mo>)</mml:mo><mml:mtext>=</mml:mtext><mml:mfrac><mml:mrow><mml:mtext>|</mml:mtext><mml:mi>t</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>T</mml:mi><mml:mo>:</mml:mo><mml:mtext>label</mml:mtext><mml:mo>(</mml:mo><mml:mi>t</mml:mi><mml:mo>)</mml:mo><mml:mtext>=</mml:mtext><mml:mi>c</mml:mi><mml:mtext>|</mml:mtext></mml:mrow><mml:mrow><mml:mtext>|</mml:mtext><mml:mi>T</mml:mi><mml:mtext>|</mml:mtext></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:math></disp-formula><p>which represents the ratio of class <inline-formula><mml:math id="ieqn3"><mml:mi>c</mml:mi></mml:math></inline-formula> elements among all data. To get the predicted class label <inline-formula><mml:math id="ieqn4"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> for a given feature vector <italic>x</italic>, <inline-formula><mml:math id="ieqn5"><mml:mi>P</mml:mi><mml:mo>(</mml:mo><mml:mi>c</mml:mi><mml:mtext>|</mml:mtext><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula> has to be calculated for each class <italic>c</italic> as given in [eq:bow_classifier], and select the highest probability as the estimated label:</p><disp-formula id="equWL7"><mml:math id="eqn6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mtext>=</mml:mtext><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:munder><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:munder><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mtext>|</mml:mtext><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>For the experiments, different metadata setups were checked: the title or the abstract itself, the title and the abstract together, and finally the title, abstract, and keywords concatenated together.</p><p>The BoW method starts with filtering and tokenization and removing stop words from text. The limit for the number of features was specified as 1000; however, this threshold was reached only when the subset including keywords was used. On limitation, features with small measured frequency are ignored.</p></sec><sec id="s2-6"><title>Baseline Analysis: Full-Text Analysis</title><p>To obtain information about EQ-5D data, the reviewers in most cases (89.0%) checked the full text. Although the aim of our study is to predict the availability of data based only on the metadata of a given publication, we have also prepared a method based on full text for comparison.</p><p>Full-text study analysis is a complex task. Based on the journals, the publishers, and even the scientific fields, the formats are widely different, with differences in markup language (HTML and TeX), format, or even plain text availability. As studies are usually available in a PDF, analyzing this format is more convenient.</p><p>The domain experts specified a list of keywords and phrases, mostly from the expressions often used in the EQ-5D descriptive system, which often indicates EQ-5D data in the study. The specified key phrases are index, value, utility/utilities, score, mobility, self-care, usual activities, pain, discomfort, anxiety, depression, looking after myself, doing usual activities, having pain or discomfort, feeling worried, feeling sad, and feeling unhappy.</p><p>To formally represent this method, let <italic>k</italic> &#x2208; <italic>K</italic> represent the abovementioned keywords, and function <inline-formula><mml:math id="ieqn6"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>d</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> represent that document <italic>d</italic> contains keyword <italic>k</italic>, where <inline-formula><mml:math id="ieqn7"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>d</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:msubsup><mml:mo>:</mml:mo><mml:mo>{</mml:mo><mml:mtext>true</mml:mtext><mml:mo>,</mml:mo><mml:mtext>false</mml:mtext><mml:mo>}</mml:mo></mml:math></inline-formula>.</p><p>A decision tree (DT) is a simple, data-driven ML method where the estimation of the trained model is explainable. A DT for binary classification can be represented as a function <inline-formula><mml:math id="ieqn8"><mml:mi>f</mml:mi><mml:mo>:</mml:mo><mml:mo>{</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:msup><mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mtext>|</mml:mtext><mml:mi>K</mml:mi><mml:mtext>|</mml:mtext></mml:mrow></mml:msup><mml:mo>&#x2192;</mml:mo><mml:mo>{</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>}</mml:mo></mml:math></inline-formula>, where <inline-formula><mml:math id="ieqn9"><mml:mtext>|</mml:mtext><mml:mi>K</mml:mi><mml:mtext>|</mml:mtext></mml:math></inline-formula> is the number of input features, in this case, the number of keywords. Each element in the feature vector can only have 2 possible values: the text either contains the keyword or not; the output of the model is also binary.</p><p>Each internal node in the DT represents a decision based on a parameter; for this case, this can be formulated as</p><disp-formula id="equWL8"><mml:math id="eqn7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>o</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>C</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing=".2em" columnspacing="1em" displaystyle="false"><mml:mtr><mml:mtd><mml:mtext>true</mml:mtext><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext>if&#x00A0;</mml:mtext><mml:msubsup><mml:mi>C</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup><mml:mtext>&#x00A0;is true</mml:mtext><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>false</mml:mtext><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext>otherwise</mml:mtext><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"/></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <italic>o</italic><sub><italic>j</italic></sub> represents the decision of a given node in the DT, where the decision is based on keyword presence represented by <inline-formula><mml:math id="ieqn10"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>d</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula>. The 2 outputs of the node represent the outcomes.</p><p>All leaf nodes in the DT define a class label, in this case either true or false. The estimated label is determined by the label associated with the leaf node reached by traversing the binary tree based on the decision rules.</p><p>During training, Gini impurity is applied during recursive partitioning to find subsets where most elements are with the same label. At each step the algorithm considers all possible splits and selects the set with optimal impurities; the process is continued recursively until further splitting does not improve the performance.</p><p>A PDF full-text processor was created in Python (Python Software Foundation), which checks every page of the provided input files for the given keywords or phrases until one is found in a given document; results are collected, and different keyword combinations are evaluated using DT.</p></sec><sec id="s2-7"><title>Selecting the Language Model, Pretraining, and Fine-Tuning</title><p>In order to select the best-performing pretrained LLM as the backbone of the filter, multiple domain-specific models were considered, detailed in <xref ref-type="table" rid="table1">Table 1</xref>. Where applicable, both cased and uncased versions were tested.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>The large language models applied in this study.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Name</td><td align="left" valign="bottom">Domain</td><td align="left" valign="bottom">Corpus</td><td align="left" valign="bottom">Sequence length</td></tr></thead><tbody><tr><td align="left" valign="top">BERT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">General</td><td align="left" valign="top">BookCorpus, Wikipedia</td><td align="left" valign="top">512</td></tr><tr><td align="left" valign="top">SciBERT<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">Scholar</td><td align="left" valign="top">Semantic Scholar</td><td align="left" valign="top">512</td></tr><tr><td align="left" valign="top">BioBERT<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">Biomedical</td><td align="left" valign="top">PubMed</td><td align="left" valign="top">512</td></tr><tr><td align="left" valign="top">BlueBERT<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">Biomedical</td><td align="left" valign="top">PubMed</td><td align="left" valign="top">512</td></tr><tr><td align="left" valign="top">SciBERT Longformer</td><td align="left" valign="top">Scholar</td><td align="left" valign="top">Untrained</td><td align="left" valign="top">4096</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table1fn2"><p><sup>b</sup>SciBERT: Scientific Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table1fn3"><p><sup>c</sup>BioBERT: Biomedical Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table1fn4"><p><sup>d</sup>BlueBERT: Biomedical Language Understanding Evaluation Bidirectional Encoder Representations from Transformers.</p></fn></table-wrap-foot></table-wrap><p>During the measurements, 2 major methods were applied to train the models:</p><list list-type="bullet"><list-item><p>Applying the pretrained language model without altering the parameters, only training the classification layer</p></list-item><list-item><p>Fine-tuning the pretrained model with a relatively small learning rate.</p></list-item></list><p>As the input consists of 3 separable texts, these were connected with the special token <inline-formula><mml:math id="ieqn11"><mml:mo>[</mml:mo><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mi>P</mml:mi><mml:mo>]</mml:mo></mml:math></inline-formula>, resulting in a single input. In case the sequence length exceeds the maximal length defined by the models, truncation was applied at the sequence end.</p><sec id="s2-7-1"><title>Applying Pretrained Models</title><p>BERT-based models are prepared to be used for text classification tasks (<xref ref-type="fig" rid="figure2">Figure 2</xref>); a classifier layer is simply attachable to the pretrained model; during training this layer is tuned, and the model parameters remain unchanged.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>A brief summary of text classification using Bidirectional Encoder Representations from Transformers (BERT)&#x2013;based large language models (LLMs). The paper metadata is concatenated, tokenized, and fed to the model as input, and finally a classification layer is built on the so-called pooled output of the network. On applying pretrained models, only the classification layer is trained; on fine-tuning, the parameters of the backbone BERT are also fitted. BERT: Bidirectional Encoder Representations from Transformers.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e86647_fig02.png"/></fig><p>To formally represent the applied method, let&#x2019;s define BERT as the pretrained language model and <inline-formula><mml:math id="ieqn12"><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mtext>BERT</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> as the parameters of the model. By having</p><disp-formula id="equWL9"><mml:math id="eqn8"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mtext>=</mml:mtext><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>BERT</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mtext>BERT</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>;</mml:mo><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn13"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> stands for the final classification layer and <inline-formula><mml:math id="ieqn14"><mml:mtext>BERT</mml:mtext><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>;</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula> represents the embeddings of <italic>x</italic> based on parameters <inline-formula><mml:math id="ieqn15"><mml:mi>&#x03B8;</mml:mi></mml:math></inline-formula>. When training, only the parameters of <inline-formula><mml:math id="ieqn16"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> are modified, represented as <inline-formula><mml:math id="ieqn17"><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>; with having <italic>L</italic> as the loss-function</p><disp-formula id="equWL10"><mml:math id="eqn9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:munder><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub></mml:munder><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mtext>|</mml:mtext><mml:mi>T</mml:mi><mml:mtext>|</mml:mtext></mml:mrow></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mtext>=</mml:mtext><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mtext>|</mml:mtext><mml:mi>T</mml:mi><mml:mtext>|</mml:mtext></mml:mrow></mml:munderover><mml:mi>L</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>BERT</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mtext>BERT</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>;</mml:mo><mml:msub><mml:mi>&#x03B8;</mml:mi><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>defines the expression to be minimized using the applied optimization algorithm, usually a gradient-descent-based technique. For binary classification, <italic>L</italic> is binary cross-entropy, where <italic>y</italic><sub><italic>i</italic></sub> represents the true binary label for <italic>t</italic><sub><italic>i</italic></sub>.</p><p>It is important to point out that the embeddings obtained from <inline-formula><mml:math id="ieqn18"><mml:mtext>BERT</mml:mtext><mml:mo>(</mml:mo><mml:mo>)</mml:mo></mml:math></inline-formula> are generated using the parameters <inline-formula><mml:math id="ieqn19"><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mtext>BERT</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>, which parameters are not updated during training, only <inline-formula><mml:math id="ieqn20"><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> are tuned.</p><p>The pretrained tokenizer is used in all cases, with application of padding to fill the maximal token length and truncation applied on exceeding it. Sparse categorical cross-entropy is applied as the loss function with Adam [<xref ref-type="bibr" rid="ref62">62</xref>] as the optimizer. Training was terminated when the loss measured on the validation dataset stopped improving, with a patience of 10 epochs. Model performance is evaluated based on the test subset.</p></sec><sec id="s2-7-2"><title>Fine-Tuning the Models</title><p>Another approach for LLM-based text classification is the fine-tuning of a pretrained model. In this case, the very same structure is applied with a classification layer at the end of the model; however, during training, the backbone model parameters are also tuned.</p><p>Let</p><disp-formula id="equWL11"><mml:math id="eqn10"><mml:mi>&#x0398;</mml:mi><mml:mtext>=</mml:mtext><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mtext>BERT</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x222A;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub></mml:math></disp-formula><p>represent all parameters of the model,</p><disp-formula id="equWL12"><mml:math id="eqn11"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mtext>=</mml:mtext><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi mathvariant="normal">&#x0398;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>gives the estimated label based on input <italic>t</italic> and all parameters. During training</p><disp-formula id="equWL13"><mml:math id="eqn12"><mml:munder><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x0398;</mml:mi></mml:mrow></mml:munder><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mtext>|</mml:mtext><mml:mi>T</mml:mi><mml:mtext>|</mml:mtext></mml:mrow></mml:mfrac><mml:mrow><mml:msubsup><mml:mo stretchy="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mtext>=</mml:mtext><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mtext>|</mml:mtext><mml:mi>T</mml:mi><mml:mtext>|</mml:mtext></mml:mrow></mml:msubsup><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mtext>classifier</mml:mtext></mml:mrow></mml:msub><mml:mo>(</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi>&#x0398;</mml:mi><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:math></disp-formula><p>is optimized.</p><p>During our experiments, we applied multiple learning rates to experience the effects on generalization: learning rate={10<sup>&#x2013;4</sup>,2&#x00D7;10<sup>&#x2013;4</sup>,5&#x00D7;10<sup>&#x2013;4</sup>,10<sup>&#x2013;5</sup><italic>,</italic>2&#x00D7;10<sup>&#x2013;5</sup>,5&#x00D7;10<sup>&#x2013;5</sup>,10<sup>&#x2013;6</sup><italic>,</italic>2&#x00D7;10<sup>&#x2013;6</sup>,5&#x00D7;10<sup>&#x2013;6</sup>}.</p><p>To reflect model stability, each fine-tuning experiment was repeated 5 times with different random seeds. These repetitions are intended as a descriptive assessment of variability across runs; reported values represent the mean and standard deviation for these results. Given the pilot scope and limited sample size, statistical analysis was not performed in this study; more extensive testing (eg, confidence intervals and significance tests) will be included in future large-scale experiments. Tokenization and training termination are handled as described previously.</p><p>A special case in this method is the SciBERT-Longformer: as the model is untrained, a regular training is applied, with the application of the SciBERT vocabulary for tokenization.</p><p>Results are evaluated using different, well-known metrics:</p><disp-formula id="equWL15"><mml:math id="eqn13"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtext>classification accuracy</mml:mtext><mml:mtext>=</mml:mtext><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mtext>+</mml:mtext><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mtext>+</mml:mtext><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equWL16"><mml:math id="eqn14"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtext>precision</mml:mtext><mml:mtext>=</mml:mtext><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mtext>+</mml:mtext><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equWL17"><mml:math id="eqn15"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtext>specificity</mml:mtext><mml:mtext>=</mml:mtext><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mi>N</mml:mi></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equWL18"><mml:math id="eqn16"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtext>sensitivity (recall)</mml:mtext><mml:mtext>=</mml:mtext><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mi>P</mml:mi></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equWL19"><mml:math id="eqn17"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mtext>F</mml:mtext><mml:mn>1</mml:mn></mml:msub><mml:mtext>-score</mml:mtext><mml:mtext>=</mml:mtext><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mtext>*</mml:mtext><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mtext>*</mml:mtext><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mtext>+</mml:mtext><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mtext>+</mml:mtext><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <italic>P</italic> and <italic>N</italic> refer to the number of positives and negatives, true positive (TP), true negative (TN), FP, and false negative (FN) refer to the number of correctly classified positives and negatives, and falsely classified positives and negatives, respectively.</p><p>For the screening-oriented interpretation of the best-performing fine-tuned model, additional measures were calculated from the held-out test confusion matrix. Positive predictive value (PPV) was defined as (<inline-formula><mml:math id="ieqn21"><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mtext>/</mml:mtext><mml:mo>(</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mtext>+</mml:mtext><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula>), negative predictive value (NPV) as (<inline-formula><mml:math id="ieqn22"><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mtext>/</mml:mtext><mml:mo>(</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mtext>+</mml:mtext><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula>), and the number of FNs was reported explicitly because missed eligible studies are critical in systematic review workflows. Workload saving was estimated as the proportion of records predicted as negative, (<inline-formula><mml:math id="ieqn23"><mml:mo>(</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mtext>+</mml:mtext><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>)</mml:mo><mml:mtext>/</mml:mtext><mml:mo>(</mml:mo><mml:mi>P</mml:mi><mml:mtext>+</mml:mtext><mml:mi>N</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula>), representing the share of records that would not require manual review if the classifier were used as an exclusion filter. This is reported only as an exploratory indicator; FNs would not be acceptable in an automated screening workflow. Wilson 95% CIs were calculated for proportion-based metrics to reflect uncertainty due to the small test set.</p></sec></sec><sec id="s2-8"><title>Ethical Considerations</title><p>This study did not involve human participants, human data, or human tissue and therefore did not require approval from an institutional review board (IRB) or informed consent procedures. No identifying personal information was collected or processed, and the work adheres to relevant guidelines and regulations, including the Committee on Publication Ethics (COPE) and the principles of the Helsinki Declaration.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Results of the Manual Review</title><p>After manual review, the number of positives is 121 (60.5%), with 79 (39.5%) negatives. The reviewers also marked those studies where checking the full text was not necessary. From the 121 positives, 22 (18.1% of positives, 11.0% of all) were marked based only on the metadata (ie, the abstract contained EQ-5D data that met our inclusion criteria), and for the remaining, the decision was made based on the full text review. For negatives, the decision was never made only on the metadata; a full study was always analyzed before the decision. Disagreements between reviewers occurred in 15 cases, but all were solved by detailed revision and discussion of the respective publications. The studies included based on abstract review (n=22) were published between 2003 and 2022, while those included based on full text review (n=99) were released between 1999 and 2022, and the excluded studies (n=79) came out between 2006 and 2022. In all the 3 manual selection subgroups, journal article (59.1%, 38.4%, and 34.2%) and research report (22.7%, 40.4%, and 32.9%) were the most frequent PubMed publication types, respectively. Although in different proportions, observational studies (9.1%, 2.0%, and 5.1%), validation studies (4.5%, 6.1%, and 5.1%), and systematic reviews (4.5%, 1%, and 5.1%) occurred in all 3 subgroups. In addition, randomized controlled trials (RCTs; n=6, 6.1%), multicenter studies (n=3, 3.0%), a review, a pragmatic clinical trial, and a book were found in the full text-based subgroup, while RCTs (n=8, 10.1%), multicenter studies (n=2, 2.5%), a review, a letter, and a twin study were found in the excluded subgroup.</p></sec><sec id="s3-2"><title>Results for BoW-Based Classification</title><p>To have a baseline architecture before applying language models, a BoW-based Na&#x00EF;ve Bayes classifier is trained and evaluated on the same subsets, results are represented in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Precision, recall, <italic>F</italic><sub>1</sub>-score, and accuracy for the bag-of-words-based Na&#x00EF;ve Bayes classifier. Results clearly indicate overfitting: perfect training performance but near-random test performance. Values are weighted by class size to remove the effects of data imbalance.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom" colspan="4">Train subset</td><td align="left" valign="bottom" colspan="4">Test subset</td></tr></thead><tbody><tr><td align="left" valign="top">Dataset</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Accuracy</td></tr><tr><td align="left" valign="top">Title</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td></tr><tr><td align="left" valign="top">Abstract</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td></tr><tr><td align="left" valign="top">Title+abstract</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td><td align="left" valign="top">0.53</td></tr><tr><td align="left" valign="top">Title+abstract+keywords</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.99</td><td align="left" valign="top">0.52</td><td align="left" valign="top">0.52</td><td align="left" valign="top">0.52</td><td align="left" valign="top">0.52</td></tr></tbody></table></table-wrap><p>It is visible that the BoW-based approach overfitted on the training data, resulting in low performance on the test subset, which is nearly random.</p></sec><sec id="s3-3"><title>Estimation of Ground Truth Using Full-Text Analysis</title><p>The human experts labeled the documents after reviewing the document metadata and the full text, providing the ground truth for our problem in predicting the label based on only the metadata. To get an estimation of the ground truth, a full text search was performed based on the keywords defined by the human experts.</p><p><xref ref-type="table" rid="table3">Table 3</xref> shows the number of occurrences for each keyword and keyphrase, along with the calculated correlation coefficient. The retrieved values show a weak connection between the occurrence of each single keyword and the expected label. If a classifier is based on these values alone, a near-random classification accuracy is the result.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Results of the full-text analysis: each keyword was checked in all 200 documents. Count refers to the total number of occurrences, and correlation is the correlation coefficient calculated for each keyword and the actual label.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Keyword</td><td align="left" valign="bottom">Count</td><td align="left" valign="bottom">Pearson correlation coefficient</td></tr></thead><tbody><tr><td align="left" valign="top">Index</td><td align="left" valign="top">150</td><td align="left" valign="top">0.24</td></tr><tr><td align="left" valign="top">Value</td><td align="left" valign="top">182</td><td align="left" valign="top">0.25</td></tr><tr><td align="left" valign="top">Utility</td><td align="left" valign="top">101</td><td align="left" valign="top">0.08</td></tr><tr><td align="left" valign="top">Utilities</td><td align="left" valign="top">47</td><td align="left" valign="top">0.13</td></tr><tr><td align="left" valign="top">Score</td><td align="left" valign="top">187</td><td align="left" valign="top">0.16</td></tr><tr><td align="left" valign="top">Mobility</td><td align="left" valign="top">136</td><td align="left" valign="top">0.19</td></tr><tr><td align="left" valign="top">Self-care</td><td align="left" valign="top">106</td><td align="left" valign="top">0.22</td></tr><tr><td align="left" valign="top">Usual activities</td><td align="left" valign="top">91</td><td align="left" valign="top">0.20</td></tr><tr><td align="left" valign="top">Pain</td><td align="left" valign="top">181</td><td align="left" valign="top">0.16</td></tr><tr><td align="left" valign="top">Discomfort</td><td align="left" valign="top">119</td><td align="left" valign="top">0.23</td></tr><tr><td align="left" valign="top">Anxiety</td><td align="left" valign="top">139</td><td align="left" valign="top">0.20</td></tr><tr><td align="left" valign="top">Depression</td><td align="left" valign="top">143</td><td align="left" valign="top">0.19</td></tr><tr><td align="left" valign="top">Looking after myself</td><td align="left" valign="top">2</td><td align="left" valign="top">0.08</td></tr><tr><td align="left" valign="top">Doing usual activities</td><td align="left" valign="top">2</td><td align="left" valign="top">0.08</td></tr><tr><td align="left" valign="top">Having pain or discomfort</td><td align="left" valign="top">2</td><td align="left" valign="top">0.08</td></tr><tr><td align="left" valign="top">Feeling worried</td><td align="left" valign="top">1</td><td align="left" valign="top">0.06</td></tr><tr><td align="left" valign="top">Feeling sad</td><td align="left" valign="top">0</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Feeling unhappy</td><td align="left" valign="top">0</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>The trained DT is visualized in <xref ref-type="fig" rid="figure3">Figure 3</xref>. The depth of the full tree is 12 levels, and the classification accuracy is 87%. It is worth mentioning that DTs with a limited depth also give similar performance: an 8-level DT scores 82.5%, and a DT with 6 levels results in a 78.5% classification accuracy. The goal of this analysis was to find the relationship between the target label and keywords and their different combinations; therefore, no test subset was selected, meaning that the results are fully fitted to the whole dataset.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Visualization of the decision tree trained on the analyzed keywords of the full-text data. For readability, only the first 3 levels of the tree are shown. Each internal node represents a decision, while leaf nodes represent the outcomes. The first line in each node shows the expression as &#x201C;k \leq 0.5&#x201D; for a given keyphrase &#x2018;k,&#x2019; which can be interpreted as &#x201C;if the document does not contain keyphrase &#x2018;k.&#x2019;&#x201D; The left subtree represents nodes where the expression is evaluated as true; the right subtree represents nodes where the evaluation is false. Other labels in the nodes include the gini impurity value, sample shows the proportion of all elements affected at a given node, and values show the mean of each value for each class.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e86647_fig03.png"/></fig><p>The results of this experiment show that the analysis of the full text is complicated, while the application of high-performing document-based classification methods is not possible without document, journal, or publisher-based preprocessing; a basic dictionary-based method fails to close on the performance of the domain experts.</p></sec><sec id="s3-4"><title>Comparing Different Pretrained Language Models</title><p>The following experiment consisted of applying the pretrained language model without altering the parameters, only training the classification layer. Results are detailed in <xref ref-type="table" rid="table4">Table 4</xref>. It is easy to conclude that while results in some cases are above the near-random BoW baseline, performance falls short of the generally accepted marks.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Performance of pretrained Bidirectional Encoder Representations from Transformers (BERT)&#x2013;family models without alterations; only the classifier parameters were trained. Reported metrics of the table include accuracy, precision, recall, and <italic>F</italic><sub>1</sub>-score for the highest performed configuration of each model and dataset (defined by top test accuracy).</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" colspan="2">Model and applied dataset</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">BERT<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title</td><td align="left" valign="top">0.63</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.63</td><td align="left" valign="top">0.54</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Abstract</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.73</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.62</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title + abstract</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.59</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title + abstract+keywords</td><td align="left" valign="top">0.61</td><td align="left" valign="top">0.76</td><td align="left" valign="top">0.61</td><td align="left" valign="top">0.47</td></tr><tr><td align="left" valign="top" colspan="2">BioBERT<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title</td><td align="left" valign="top">0.65</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.65</td><td align="left" valign="top">0.59</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Abstract</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.63</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title +abstract</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.65</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.63</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title + abstract+keywords</td><td align="left" valign="top">0.62</td><td align="left" valign="top">0.6</td><td align="left" valign="top">0.62</td><td align="left" valign="top">0.59</td></tr><tr><td align="left" valign="top" colspan="2">SciBERT<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.63</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Abstract</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.78</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.53</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title+abstract</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.57</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title+abstract+keywords</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.57</td></tr><tr><td align="left" valign="top" colspan="2">BlueBERT<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.78</td><td align="left" valign="top">0.64</td><td align="left" valign="top">0.53</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Abstract</td><td align="left" valign="top">0.63</td><td align="left" valign="top">0.77</td><td align="left" valign="top">0.63</td><td align="left" valign="top">0.51</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title+abstract</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.59</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Title+abstract+keywords</td><td align="left" valign="top">0.62</td><td align="left" valign="top">0.77</td><td align="left" valign="top">0.62</td><td align="left" valign="top">0.49</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table4fn2"><p><sup>b</sup>BioBERT: Biomedical Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table4fn3"><p><sup>c</sup>SciBERT: Scientific Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table4fn4"><p><sup>d</sup>BlueBERT: Biomedical Language Understanding Evaluation Bidirectional Encoder Representations from Transformers.</p></fn></table-wrap-foot></table-wrap><p>Experiments were done in a multiple central processing units environment, while graphics processing unit acceleration is also possible. Popular frameworks (<italic>TensorFlow</italic>, <italic>transformers</italic>, and <italic>sklearn</italic>) were applied in implementation with generic hyperparameters; details are available in the source codes that are made publicly available.</p><p>It is also trivial that when only the title or only the abstract is submitted, it is outperformed by solutions with more input data.</p></sec><sec id="s3-5"><title>Fine-Tuning the Language Models</title><p>When applying the LLMs in a transfer learning scenario, where pretrained values are used as initial parameters and are allowed to change during training, it is empirically determined that performance is higher.</p><p>In these experiments, we used the dataset where the title, the abstract, and the keywords are concatenated into a single sequence.</p><p>Results show (<xref ref-type="table" rid="table5">Table 5</xref>) that the top-performing model was the BioBERT on a specific learning rate, which slightly outperformed the BERT and the Longformer models.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>The results of fine-tuning different large language models (LLMs) with varying learning rates, sorted by classification accuracy, in descending order. All trainings were done 5 times, average accuracy, and the SD is also presented next to precision, recall, and <italic>F</italic><sub>1</sub>-score measured on the test subset.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Model</td><td align="left" valign="top">Learning rate</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top">Accuracy, mean (SD)</td><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">BERT<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">0.000002</td><td align="left" valign="top">0.70</td><td align="left" valign="top">0.67 (0.02)</td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.70</td><td align="left" valign="top">0.68</td></tr><tr><td align="left" valign="top">BioBERT<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="top">0.000050</td><td align="left" valign="top">0.70</td><td align="left" valign="top">0.66 (0.02)</td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.70</td><td align="left" valign="top">0.68</td></tr><tr><td align="left" valign="top">BERT</td><td align="left" valign="top">0.000001</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.67 (0.02)</td><td align="left" valign="top">0.74</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.64</td></tr><tr><td align="left" valign="top">BERT</td><td align="left" valign="top">0.000100</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.64 (0.03)</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.69</td></tr><tr><td align="left" valign="top">BERT</td><td align="left" valign="top">0.000010</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.67 (0.01)</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.67</td></tr><tr><td align="left" valign="top">BlueBERT<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td><td align="left" valign="top">0.000100</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.60 (0.07)</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.69</td></tr><tr><td align="left" valign="top">SciBERT<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup> Long</td><td align="left" valign="top">0.000100</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.66 (0.02)</td><td align="left" valign="top">0.76</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.63</td></tr><tr><td align="left" valign="top">BioBERT</td><td align="left" valign="top">0.000005</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.65 (0.03)</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.66</td></tr><tr><td align="left" valign="top">SciBERT</td><td align="left" valign="top">0.000100</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.64 (0.02)</td><td align="left" valign="top">0.70</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.64</td></tr><tr><td align="left" valign="top">BlueBERT</td><td align="left" valign="top">0.000005</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.62 (0.04)</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.68</td><td align="left" valign="top">0.67</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table5fn2"><p><sup>b</sup>BioBERT: Biomedical Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table5fn3"><p><sup>c</sup>BlueBERT: Biomedical Language Understanding Evaluation Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table5fn4"><p><sup>d</sup>SciBERT: Scientific Bidirectional Encoder Representations from Transformers.</p></fn></table-wrap-foot></table-wrap><p>The screening-oriented metrics (<xref ref-type="table" rid="table6">Table 6</xref>) show that the best-performing fine-tuned configuration achieved high sensitivity but low specificity. This indicates that the model was able to identify most eligible EQ-5D data reports, but at the cost of a substantial number of FPs. The false-negative count remained nonnegligible; therefore, the model should not be interpreted as a safe autonomous exclusion tool in its current form; instead, these results support early technical feasibility.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Screening-oriented performance of the best-performing fine-tuned Bidirectional Encoder Representations from Transformers (BERT) configuration on the held-out test set. Wilson 95% CIs are reported for proportion-based metrics. The confusion matrix was true positive (TP)=54, true negative (TN)=16, false positive (FP)=24, and false negative (FN)=6.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Value</td><td align="left" valign="bottom">Wilson 95% CI</td></tr></thead><tbody><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">70.0%</td><td align="left" valign="top">60.4%&#x2010;78.1%</td></tr><tr><td align="left" valign="top">Sensitivity</td><td align="left" valign="top">90.0%</td><td align="left" valign="top">79.9%&#x2010;95.3%</td></tr><tr><td align="left" valign="top">Specificity</td><td align="left" valign="top">40.0%</td><td align="left" valign="top">26.3%&#x2010;55.4%</td></tr><tr><td align="left" valign="top">PPV<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup></td><td align="left" valign="top">69.2%</td><td align="left" valign="top">58.3%&#x2010;78.4%</td></tr><tr><td align="left" valign="top">NPV<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup></td><td align="left" valign="top">72.7%</td><td align="left" valign="top">51.8%&#x2010;86.8%</td></tr><tr><td align="left" valign="top">False negatives</td><td align="left" valign="top">6</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup></td></tr><tr><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.783</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>PPV: positive predictive value.</p></fn><fn id="table6fn2"><p><sup>b</sup>NPV: negative predictive value.</p></fn><fn id="table6fn3"><p><sup>c</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>A supplementary 10-run stability analysis of the same learning-rate configuration showed test accuracy ranging from 62.0% to 70.0% (mean 66.3%, SD 2.4) and FNs ranging from 2 to 11 across seeds.</p><p>The test set contains 7 positive records where the reviewers were able to decide based only on the metadata; the model classified all of them correctly. It is worth mentioning that the training of the SciBERT longformer does not classify as fine-tuning, as the model parameters were trained from scratch. It is also visible that the BlueBERT and SciBERT models were only slightly outperformed; however, learning rate selection has a significant effect on the outcome.</p><p>It is worth noting that several model configurations achieved similar performance; however, small differences in accuracy between the top models should be interpreted cautiously. The observed performance may change under alternative train/validation/test splits or under additional training repetitions. Therefore, this reported ordering of high-performing models is presented primarily as preliminary feasibility evidence.</p><p>To ensure comparability across models, we report accuracy, precision, recall, specificity, and <italic>F</italic><sub>1</sub>-score consistently in all experiments. Future work will also include statistical analysis (eg, CIs and significance testing) to better assess model differences.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>We analyzed LLM-based methods to select EQ-5D data reports (either the EQ-5D index score or results of the EQ-5D descriptive system) in manuscripts in PubMed based on the publicly available metadata such as the title, abstract, and keywords. In the experiments, we tested the performance of multiple domain-specific pretrained models, with different architectures and setups. We found that top performance is achieved by applying fine-tuning to the models. We also found that estimating the ground truth from the full text has limitations, further emphasizing the importance of study metadata-based methods.</p><p>According to our best knowledge, this is the first study aiming to identify studies reporting EQ-5D data using the LLM-based automation method. Therefore, direct comparisons with the international literature are not feasible. T&#x00F3;th et al [<xref ref-type="bibr" rid="ref3">3</xref>] in their recent review have identified in PubMed 108 studies on SLR automation methods and a further 15 SLRs that applied automation partly or throughout the SLR process. Automation of record screening (based on title and abstract) was the most frequently studied SLR stage, and, similarly to our approach, typically manually screened publications were used for the training of the ML classifier. The BERT model and other LMs were applied in recent studies, presenting the increasing use of novel NLP techniques. Torre-Lopez et al [<xref ref-type="bibr" rid="ref63">63</xref>] also reported that interest is clearly increasing in the field of SLR automation. Sundaram and Berleant [<xref ref-type="bibr" rid="ref64">64</xref>] highlighted that there are gaps in the application of different text mining and NLP methods in different areas of the automated SLR process. Hasny et al [<xref ref-type="bibr" rid="ref65">65</xref>] applied different BERT models for paper screening and reported a significant reduction in necessary human workload.</p><p>For the interpretation of our results, some limitations of our study have to be noted. The most significant limitation is definitely the small sample of publications that we used to train and test our model. Nonetheless, this first experimental study aimed to identify methodological challenges, such as feasibility, and formulate some points to consider for which a smaller study was deemed sufficient. Our future plans include the extension of the data using semisupervised methods based on the current models, which will hopefully lower the time cost per study.</p><p>Another related limitation is that the results are based on a single, predefined split of the small dataset; therefore, estimates are split-dependent and do not support strong generalization claims. Repeated cross-validation and/or external validation on independent samples is required to assess reliability and stability. Although sensitivity was relatively high in the best-performing configuration, specificity remained limited, and FNs were still observed. Therefore, the model is not suitable for unsupervised exclusion of records in systematic review workflows.</p><p>We would like to highlight some aspects of our study that we think are worthwhile for future research in the field.</p><p>First, we did not restrict our PubMed search to specific publication or study types but considered all hits (eg, books, reviews, and twin studies, etc). Given that different types of publications and studies have different reporting designs and standards, we assume that our model would have yielded different (presumably more precise) results if we had investigated automation, for instance, only among RCTs. RCTs are one of the most frequently searched publications for health technology assessment, as RCTs provide the highest level of clinical evidence. Reporting standards for RCTs have been established and are being updated [<xref ref-type="bibr" rid="ref66">66</xref>], thus can be used as a practical gold standard for the authors on how to write the studies and to consider these reporting requirements already at the designing phase of the trials. With regard to outcomes, CONSORT (Consolidated Standards of Reporting Trials) requires to report &#x201C;Completely defined prespecified primary and secondary outcome measures, including how and when they were assessed&#x201D; (Methods section), and &#x201C;For each primary and secondary outcome, results for each group, and the estimated effect size and its precision (such as 95% CI)&#x201D; (Results section). Moreover, the CONSORT extension [<xref ref-type="bibr" rid="ref67">67</xref>] for abstracts points out that primary outcomes and their results should be reported in the abstract of the study. Therefore, we think it would be an interesting avenue for further research to test our automation method on specific study types, and RCTs could be one of the firsts, given their importance and the relatively high-quality publication practice.</p><p>Second, many publications (eg, some RCTs but more probably the observational studies, real-world data reports, etc) still do not follow the reporting standards and do not describe all important details in a complete and transparent way. Although the development of LLMs alone may bring some increase in the precision of literature review automation, we believe that no breakthrough can be achieved without improving broadly the reporting practice. This latter is a huge responsibility of authors and probably even more so of journal editors.</p><p>Third, we also believe that developers of reporting guidelines should pay more attention to the high need for automation of systematic reviews and consider the capabilities and needs of AI-based automated search, selection, and data extraction technologies. For instance, text mining and data extraction from tables and figures are much more challenging, or even not feasible for LLMs when compared to processing plain text. According to our best knowledge, this aspect has not been considered in reporting guidelines up to now. The EQUATOR (Enhancing the Quality and Transparency of health Research) Network [<xref ref-type="bibr" rid="ref68">68</xref>] gives recommendations for reporting guideline development and emphasizes the importance of inviting stakeholders with a wide range of expertise in the development group. These usually include journal editors, statisticians, epidemiologists, clinicians, but the potential role of IT experts in natural language processing has not come into focus so far.</p><p>Fourth, the EQ-5D outcome measurement tool was in the center of our study and the publication pool (out of which we randomly selected our study sample of 200 records) involved all pieces since its inception in 1990. The EQ-5D is somewhat an exceptional outcome measurement tool compared to others due to its more than 3 decades of history and, perhaps more importantly, being backed by a professional team of researchers, the EuroQol Group. Significant progress around the EQ-5D use and data sharing has been seen in the past years. For instance, the EQ-5D terminology has been established (and is being maintained), a PubMed search strategy has been developed, recommendations on how to analyze and report EQ-5D data have been published, and educational materials and user guides have been launched in various languages. Hence, the generalizability of our findings to reviews of other outcome measures with a more modest literature background and reporting standards requires further investigation. The exclusion of EQ VAS-only studies should also be considered when interpreting the results: this choice reflects the operational target of this study, namely the identification of publications reporting EQ-5D descriptive-system or index-score data. Nonetheless, it would be worth investigating (on a larger sample of EQ-5D publications) whether the more recent studies (eg, from the past 10 years) are in fact more &#x201C;readable&#x201D; for LLMs than the older ones. Fifth, we find it important to highlight that our methods and results may not be directly applicable to searches in other databases. Literature databases (eg, Embase and Scopus) may differ in many respects, that is, the list of journals indexed and the types of publications recorded, and also how they handle metadata for non-English language publications. All these may have implications on the performance of the LLMs in selecting publications. We encourage researchers to adapt and put our approach to the test in other biomedical literature databases.</p></sec><sec id="s4-2"><title>Conclusion</title><p>This study demonstrates the feasibility of automating the identification of EQ-5D data reports in PubMed based on scientific publication metadata using LLMs. Although accuracy remains moderate due to the limited dataset size, the approach successfully reproduces manual screening trends and establishes a reproducible framework for future large-scale and semisupervised analyses. It is important to point out that the current model is not applicable for autonomous screening or unsupervised exclusion of records; this study was designed as a proof-of-concept evaluation of whether EQ-5D data reporting in the full text of biomedical publications can be predicted from PubMed metadata alone.</p><p>As the results of this study show moderate performance, it itself is not applicable for autonomous screening; however, it provides a foundation for future studies using larger datasets.</p><p>To efficiently increase the training data, we aim to design a semisupervised approach, where reviewers are supported by the estimations from our model, along with a confidence value. We believe that by sampling the available data considering the estimations and certainty, a process of reviewing could be designed, where time-consuming is minimized.</p><p>The current model is not ready for automated screening or unsupervised exclusion of records. Its potential use is to support human reviewers by prioritizing records or flagging likely candidates for further assessment.</p><p>With an increased number of training samples, different models might outperform the currently used fine-tuned BioBERT; pretrained models or an untrained architecture (eg, the SciBERT Longformer) could achieve a higher performance. Other architectures, multiheaded BERT structures, or ensemble classifiers based on multiple trained models could also be used to improve performance. Robustness and generalizability require validation using external datasets in future work.</p></sec></sec></body><back><ack><p>This project has been supported by the National Research, Development, and Innovation Fund of Hungary, financed under the TKP2021-NKTA-36 funding scheme (Project: Development and evaluation of innovative and digital health technologies; Subproject: Evaluation of digital medical devices: efficacy, safety, and social utility).</p><p>The authors declare the use of generative AI (GenAI) in the research and writing process. According to the GAIDeT taxonomy (2025), the following tasks were delegated to GenAI tools under full human supervision:</p><p>- Proofreading and editing</p><p>- Reformatting</p><p>The GenAI tool used was: ChatGPT 5.4.</p><p>Responsibility for the final manuscript lies entirely with the authors.</p><p>GenAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p><p>Declaration submitted by: Collective responsibility</p></ack><notes><sec><title>Funding</title><p>This project has been supported by the National Research, Development, and Innovation Fund of Hungary, financed under the TKP2021-NKTA-36 funding scheme (Project: Development and evaluation of innovative and digital health technologies; Subproject: Evaluation of digital medical devices: efficacy, safety and social utility).</p></sec><sec><title>Data Availability</title><p>The datasets generated or analyzed during this study and the source code used for the experiments are available in the GitHub repository [<xref ref-type="bibr" rid="ref69">69</xref>]. The dataset is available in the &#x201C;data&#x201D; directory of the repository. Full-text publications are not included because of publisher copyright restrictions; it is recommended to collect them based on the respective DOIs.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: GK, ZZ, LG, MP</p><p>Data curation: GK, JTC, ZZ, MP</p><p>Formal analysis: GK, ZZ, MP</p><p>Funding acquisition: ZZ, LG, MP</p><p>Investigation: ZZ, MP</p><p>Methodology: GK, JTC, ZZ, MP</p><p>Project administration: JTC, MP</p><p>Resources: GK, JTC, LG, MP</p><p>Supervision: MP</p><p>Validation: GK, JTC, ZZ, LG, MP</p><p>Visualization: GK, MP</p><p>Writing &#x2013; original draft: GK, JTC, MP</p><p>Writing &#x2013; review &#x0026; editing: GK, ZZ, LG, MP</p><p>Software: GK, JTC</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb2">biLM</term><def><p>bidirectional language model</p></def></def-item><def-item><term id="abb3">BioBERT</term><def><p>Biomedical Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb4">BlueBERT</term><def><p>Biomedical Language Understanding Evaluation Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb5">BoW</term><def><p>bag-of-words</p></def></def-item><def-item><term id="abb6">CONSORT</term><def><p>Consolidated Standards of Reporting Trials</p></def></def-item><def-item><term id="abb7">COPE</term><def><p>Committee on Publication Ethics</p></def></def-item><def-item><term id="abb8">DT</term><def><p>decision tree</p></def></def-item><def-item><term id="abb9">EQ VAS</term><def><p>EuroQol Visual Analogue Scale</p></def></def-item><def-item><term id="abb10">EQUATOR</term><def><p>Enhancing the Quality and Transparency of health Research</p></def></def-item><def-item><term id="abb11">FN</term><def><p>false negative</p></def></def-item><def-item><term id="abb12">FP</term><def><p>false positive</p></def></def-item><def-item><term id="abb13">HRQoL</term><def><p>health-related quality of life</p></def></def-item><def-item><term id="abb14">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb15">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb16">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb17">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb18">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb19">PPV</term><def><p>positive predictive value</p></def></def-item><def-item><term id="abb20">PROM</term><def><p>patient-reported outcome measure</p></def></def-item><def-item><term id="abb21">QALY</term><def><p>quality-adjusted life year</p></def></def-item><def-item><term id="abb22">RCT</term><def><p>randomized controlled trial</p></def></def-item><def-item><term id="abb23">SciBERT</term><def><p>Scientific Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb24">SLR</term><def><p>systematic literature review</p></def></def-item><def-item><term id="abb25">TF-IDF</term><def><p>Term Frequency-Inverse Document Frequency</p></def></def-item><def-item><term id="abb26">TN</term><def><p>true negative</p></def></def-item><def-item><term id="abb27">TP</term><def><p>true positive</p></def></def-item><def-item><term id="abb28">VAS</term><def><p>visual analog scale</p></def></def-item><def-item><term id="abb29">WSS@95</term><def><p>workload savings at 95% recall</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Cochrane handbook for systematic reviews of interventions (current version)</article-title><source>Cochran</source><access-date>2026-08-20</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.cochrane.org/authors/handbooks-and-manuals/handbook/current">https://www.cochrane.org/authors/handbooks-and-manuals/handbook/current</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zah</surname><given-names>V</given-names> </name><name name-style="western"><surname>Burrell</surname><given-names>A</given-names> </name><name name-style="western"><surname>Asche</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zrubka</surname><given-names>Z</given-names> </name></person-group><article-title>Paying for digital health interventions &#x2013; what evidence is needed?</article-title><source>Acta Polytech Hung</source><year>2022</year><volume>19</volume><issue>9</issue><fpage>179</fpage><lpage>199</lpage><pub-id pub-id-type="doi">10.12700/APH.19.9.2022.9.10</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>T&#x00F3;th</surname><given-names>B</given-names> </name><name name-style="western"><surname>Berek</surname><given-names>L</given-names> </name><name name-style="western"><surname>Gul&#x00E1;csi</surname><given-names>L</given-names> </name><name name-style="western"><surname>P&#x00E9;ntek</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zrubka</surname><given-names>Z</given-names> </name></person-group><article-title>Automation of systematic reviews of biomedical literature: a scoping review of studies indexed in PubMed</article-title><source>Syst Rev</source><year>2024</year><month>07</month><day>8</day><volume>13</volume><issue>1</issue><fpage>174</fpage><pub-id pub-id-type="doi">10.1186/s13643-024-02592-3</pub-id><pub-id pub-id-type="medline">38978132</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blaizot</surname><given-names>A</given-names> </name><name name-style="western"><surname>Veettil</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Saidoung</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Using artificial intelligence methods for systematic review in health sciences: a systematic review</article-title><source>Res Synth Methods</source><year>2022</year><month>05</month><volume>13</volume><issue>3</issue><fpage>353</fpage><lpage>362</lpage><pub-id pub-id-type="doi">10.1002/jrsm.1553</pub-id><pub-id pub-id-type="medline">35174972</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kowsari</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jafari Meimandi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Heidarysafa</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mendu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Barnes</surname><given-names>L</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>D</given-names> </name></person-group><article-title>Text classification algorithms: a survey</article-title><source>Information</source><year>2019</year><volume>10</volume><issue>4</issue><fpage>150</fpage><pub-id pub-id-type="doi">10.3390/info10040150</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>LeCun</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bengio</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hinton</surname><given-names>G</given-names> </name></person-group><article-title>Deep learning</article-title><source>Nature</source><year>2015</year><month>05</month><day>28</day><volume>521</volume><issue>7553</issue><fpage>436</fpage><lpage>444</lpage><pub-id pub-id-type="doi">10.1038/nature14539</pub-id><pub-id pub-id-type="medline">26017442</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Brants</surname><given-names>T</given-names> </name><name name-style="western"><surname>Popat</surname><given-names>A</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Och</surname><given-names>FJ</given-names> </name><name name-style="western"><surname>Dean</surname><given-names>J</given-names> </name></person-group><article-title>Large language models in machine translation</article-title><access-date>2026-08-20</access-date><conf-name>Proceedings of the 2007 joint conference on empirical methods in natural language processing and computational natural language learning (EMNLP-CoNLL)</conf-name><conf-date>Jun 28-30, 2007</conf-date><conf-loc>Prague, Czech Republic</conf-loc><fpage>858</fpage><lpage>867</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/volumes/D07-1/">https://aclanthology.org/volumes/D07-1/</ext-link></comment></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Floridi</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chiriatti</surname><given-names>M</given-names> </name></person-group><article-title>GPT-3: its nature, scope, limits, and consequences</article-title><source>Minds Mach</source><year>2020</year><month>12</month><volume>30</volume><issue>4</issue><fpage>681</fpage><lpage>694</lpage><pub-id pub-id-type="doi">10.1007/s11023-020-09548-1</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Min</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ross</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sulem</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Recent advances in natural language processing via large pre-trained language models: a survey</article-title><source>ACM Comput Surv</source><year>2024</year><month>02</month><day>29</day><volume>56</volume><issue>2</issue><fpage>1</fpage><lpage>40</lpage><pub-id pub-id-type="doi">10.1145/3605943</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>HaCohen-Kerner</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yigal</surname><given-names>Y</given-names> </name></person-group><article-title>The influence of preprocessing on text classification using a bag-of-words representation</article-title><source>PLoS One</source><year>2020</year><volume>15</volume><issue>5</issue><fpage>e0232525</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0232525</pub-id><pub-id pub-id-type="medline">32357164</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Selva Birunda</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kanniga Devi</surname><given-names>R</given-names> </name></person-group><article-title>A review on word embedding techniques for text classification</article-title><conf-name>Innovative Data Communication Technologies and Application: Proceedings of ICIDCA 2020</conf-name><conf-date>Sep 3-4, 2020</conf-date><conf-loc>Coimbatore, India</conf-loc><fpage>267</fpage><lpage>281</lpage><pub-id pub-id-type="doi">10.1007/978-981-15-9651-3_23</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Peters</surname><given-names>M</given-names> </name><name name-style="western"><surname>Neumann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zettlemoyer</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yih</surname><given-names>W tau</given-names> </name></person-group><article-title>Dissecting contextual word embeddings: architecture and representation</article-title><access-date>2026-08-01</access-date><conf-name>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Oct 31 to Nov 4, 2018</conf-date><conf-loc>Brussels, Belgium</conf-loc><fpage>1499</fpage><lpage>1509</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://aclweb.org/anthology/D18-1">http://aclweb.org/anthology/D18-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D18-1179</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Devlin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Toutanova</surname><given-names>K</given-names> </name></person-group><article-title>Bert: pre-training of deep bidirectional transformers for language understanding</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2018</comment><pub-id pub-id-type="doi">10.48550/arXiv.1810.04805</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Child</surname><given-names>R</given-names> </name><name name-style="western"><surname>Luan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Amodei</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sutskever</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Language models are unsupervised multitask learners</article-title><year>2019</year><access-date>2026-08-26</access-date><publisher-name>OpenAI</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://cdn.openai.com/better-language-models/language_models_are_unsupervised_multitask_learners.pdf">https://cdn.openai.com/better-language-models/language_models_are_unsupervised_multitask_learners.pdf</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>N</given-names> </name><name name-style="western"><surname>Uszkoreit</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>L</given-names> </name><name name-style="western"><surname>Gomez</surname><given-names>AN</given-names> </name><etal/></person-group><article-title>Attention is all you need</article-title><source>Advances in Neural Information Processing Systems 30</source><year>2017</year><access-date>2026-08-26</access-date><publisher-name>Curran Associates</publisher-name><fpage>5998</fpage><lpage>6008</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf">https://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meadows</surname><given-names>KA</given-names> </name></person-group><article-title>Patient-reported outcome measures: an overview</article-title><source>Br J Community Nurs</source><year>2011</year><month>03</month><volume>16</volume><issue>3</issue><fpage>146</fpage><lpage>151</lpage><pub-id pub-id-type="doi">10.12968/bjcn.2011.16.3.146</pub-id><pub-id pub-id-type="medline">21378658</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>Search for EQ-5D documents</article-title><source>EuroQol</source><access-date>2026-08-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://eq-5dpublications.euroqol.org/">https://eq-5dpublications.euroqol.org/</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Devlin</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Brooks</surname><given-names>R</given-names> </name></person-group><article-title>EQ-5D and the EuroQol group: past, present and future</article-title><source>Appl Health Econ Health Policy</source><year>2017</year><month>04</month><volume>15</volume><issue>2</issue><fpage>127</fpage><lpage>137</lpage><pub-id pub-id-type="doi">10.1007/s40258-017-0310-5</pub-id><pub-id pub-id-type="medline">28194657</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Longworth</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Young</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Use of generic and condition-specific measures of health-related quality of life in NICE decision-making: a systematic review, statistical modelling and survey</article-title><source>Health Technol Assess</source><year>2014</year><volume>18</volume><issue>9</issue><pub-id pub-id-type="doi">10.3310/hta18090</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>White</surname><given-names>J</given-names> </name></person-group><article-title>PubMed 2.0</article-title><source>Med Ref Serv Q</source><year>2020</year><volume>39</volume><issue>4</issue><fpage>382</fpage><lpage>387</lpage><pub-id pub-id-type="doi">10.1080/02763869.2020.1826228</pub-id><pub-id pub-id-type="medline">33085945</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gonz&#x00E1;lez-Carvajal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Garrido-Merch&#x00E1;n</surname><given-names>EC</given-names> </name></person-group><article-title>Comparing BERT against traditional machine learning text classification</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 12, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2005.13012</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gasparetto</surname><given-names>A</given-names> </name><name name-style="western"><surname>Marcuzzo</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zangari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Albarelli</surname><given-names>A</given-names> </name></person-group><article-title>A survey on text classification algorithms: from text to predictions</article-title><source>Information</source><year>2022</year><volume>13</volume><issue>2</issue><fpage>83</fpage><pub-id pub-id-type="doi">10.3390/info13020083</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dogra</surname><given-names>V</given-names> </name><name name-style="western"><surname>Verma</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A complete process of text classification system using state-of-the-art NLP models</article-title><source>Comput Intell Neurosci</source><year>2022</year><volume>2022</volume><fpage>1883698</fpage><pub-id pub-id-type="doi">10.1155/2022/1883698</pub-id><pub-id pub-id-type="medline">35720939</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A survey on text classification: from traditional to deep learning</article-title><source>ACM Trans Intell Syst Technol</source><year>2022</year><month>04</month><day>30</day><volume>13</volume><issue>2</issue><fpage>1</fpage><lpage>41</lpage><pub-id pub-id-type="doi">10.1145/3495162</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>C</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><etal/></person-group><article-title>A comprehensive survey on pretrained foundation models: a history from BERT to ChatGPT</article-title><source>Int J Mach Learn Cybern</source><year>2025</year><month>12</month><volume>16</volume><issue>12</issue><fpage>9851</fpage><lpage>9915</lpage><pub-id pub-id-type="doi">10.1007/s13042-024-02443-6</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alammary</surname><given-names>AS</given-names> </name></person-group><article-title>BERT models for Arabic text classification: a systematic review</article-title><source>Applied Sciences</source><year>2022</year><volume>12</volume><issue>11</issue><fpage>5720</fpage><pub-id pub-id-type="doi">10.3390/app12115720</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Almaliki</surname><given-names>M</given-names> </name><name name-style="western"><surname>Almars</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Gad</surname><given-names>I</given-names> </name><name name-style="western"><surname>Atlam</surname><given-names>ES</given-names> </name></person-group><article-title>ABMM: Arabic BERT-mini model for hate-speech detection on social media</article-title><source>Electronics (Basel)</source><year>2023</year><volume>12</volume><issue>4</issue><fpage>1048</fpage><pub-id pub-id-type="doi">10.3390/electronics12041048</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Automatic text classification of actionable radiology reports of tinnitus patients using bidirectional encoder representations from transformer (BERT) and in-domain pre-training (IDPT)</article-title><source>BMC Med Inform Decis Mak</source><year>2022</year><month>07</month><day>30</day><volume>22</volume><issue>1</issue><fpage>200</fpage><pub-id pub-id-type="doi">10.1186/s12911-022-01946-y</pub-id><pub-id pub-id-type="medline">35907966</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osv&#x00E1;th</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>ZG</given-names> </name><name name-style="western"><surname>K&#x00F3;sa</surname><given-names>K</given-names> </name></person-group><article-title>Analyzing narratives of patient experiences: a BERT topic modeling approach</article-title><source>Acta Polytech Hung</source><year>2023</year><volume>20</volume><issue>7</issue><fpage>153</fpage><lpage>171</lpage><pub-id pub-id-type="doi">10.12700/APH.20.7.2023.7.9</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Hsiang</surname><given-names>J</given-names> </name></person-group><article-title>Patent classification by fine-tuning BERT language model</article-title><source>World Pat Inf</source><year>2020</year><month>06</month><volume>61</volume><fpage>101965</fpage><pub-id pub-id-type="doi">10.1016/j.wpi.2020.101965</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>C</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name></person-group><article-title>How to fine-tune bert for text classification? chinese computational linguistics</article-title><conf-name>18th China National Conference on Chinese Computational Linguistics (CCL 2019)</conf-name><conf-date>Oct 18-20, 2019</conf-date><conf-loc>Kunming, China</conf-loc><fpage>194</fpage><lpage>206</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-32381-3_16</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Adhikari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ram</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>J</given-names> </name></person-group><article-title>DocBERT: BERT for document classification</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 22, 2019</comment><pub-id pub-id-type="doi">10.48550/arXiv.1904.08398</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>M</given-names> </name></person-group><article-title>A new method of improving bert for text classification</article-title><conf-name>9th International Conference on Intelligent Science and Big Data Engineering (IScIDE 2019)</conf-name><conf-date>Oct 17-20, 2019</conf-date><conf-loc>Nanjing, China</conf-loc><fpage>442</fpage><lpage>452</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-36204-1_37</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kishore</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Weinberger</surname><given-names>KQ</given-names> </name><name name-style="western"><surname>Artzi</surname><given-names>Y</given-names> </name></person-group><article-title>Bertscore: evaluating text generation with bert</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 24, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.1904.09675</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name></person-group><article-title>Sentence-BERT: sentence embeddings using siamese BERT-networks</article-title><year>2019</year><month>08</month><day>27</day><access-date>2026-08-26</access-date><conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-date>Nov 3, 2019 to Nov 7, 2029</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.aclweb.org/anthology/D19-1">https://www.aclweb.org/anthology/D19-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D19-1410</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khadhraoui</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bellaaj</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ammar</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Hamam</surname><given-names>H</given-names> </name><name name-style="western"><surname>Jmaiel</surname><given-names>M</given-names> </name></person-group><article-title>Survey of BERT-base models for scientific text classification: COVID-19 case study</article-title><source>Applied Sciences</source><year>2022</year><volume>12</volume><issue>6</issue><fpage>2891</fpage><pub-id pub-id-type="doi">10.3390/app12062891</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Beltagy</surname><given-names>I</given-names> </name><name name-style="western"><surname>Lo</surname><given-names>K</given-names> </name><name name-style="western"><surname>Cohan</surname><given-names>A</given-names> </name></person-group><article-title>SciBERT: a pretrained language model for scientific text</article-title><access-date>2026-08-01</access-date><conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-date>Nov 3, 2019 to Nov 7, 2029</conf-date><conf-loc>Hong Kong, China</conf-loc><fpage>3615</fpage><lpage>3620</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.aclweb.org/anthology/D19-1">https://www.aclweb.org/anthology/D19-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D19-1371</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><etal/></person-group><article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title><source>Bioinformatics</source><year>2020</year><month>02</month><day>15</day><volume>36</volume><issue>4</issue><fpage>1234</fpage><lpage>1240</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id><pub-id pub-id-type="medline">31501885</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Peng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name></person-group><article-title>Transfer learning in biomedical natural language processing: an evaluation of BERT and elmo on ten benchmarking datasets</article-title><conf-name>Proceedings of the 18th BioNLP Workshop and Shared Task</conf-name><conf-date>2019</conf-date><conf-loc>Florence, Italy</conf-loc><fpage>58</fpage><lpage>65</lpage><pub-id pub-id-type="doi">10.18653/v1/W19-5006</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Beltagy</surname><given-names>I</given-names> </name><name name-style="western"><surname>Peters</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Cohan</surname><given-names>A</given-names> </name></person-group><article-title>Longformer: the long-document transformer</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 10, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2004.05150</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Mutasodirin</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Prasojo</surname><given-names>RE</given-names> </name></person-group><article-title>Investigating text shortening strategy in BERT: truncation vs summarization</article-title><source>2021 International Conference on Advanced Computer Science and Information Systems (ICACSIS</source><year>2021</year><publisher-name>IEEE</publisher-name><fpage>1</fpage><lpage>5</lpage><pub-id pub-id-type="doi">10.1109/ICACSIS53237.2021.9631364</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Su</surname><given-names>J</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>D</given-names> </name></person-group><article-title>Improving BERT-based text classification with auxiliary sentence and domain knowledge</article-title><source>IEEE Access</source><year>2019</year><volume>7</volume><fpage>176600</fpage><lpage>176612</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2019.2953990</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Pei</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Pre-trained language models in biomedical domain: a systematic survey</article-title><source>ACM Comput Surv</source><year>2024</year><month>03</month><day>31</day><volume>56</volume><issue>3</issue><fpage>1</fpage><lpage>52</lpage><pub-id pub-id-type="doi">10.1145/3611651</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mosbach</surname><given-names>M</given-names> </name><name name-style="western"><surname>Andriushchenko</surname><given-names>M</given-names> </name><name name-style="western"><surname>Klakow</surname><given-names>D</given-names> </name></person-group><article-title>On the stability of fine-tuning BERT: misconceptions, explanations, and strong baselines</article-title><access-date>2026-08-26</access-date><conf-name>9th international conference on learning representations</conf-name><conf-date>May 3-7, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=nzpLWnVAyah">https://openreview.net/forum?id=nzpLWnVAyah</ext-link></comment></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gates</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gates</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sebastianski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Guitard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Elliott</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Hartling</surname><given-names>L</given-names> </name></person-group><article-title>The semi-automation of title and abstract screening: a retrospective exploration of ways to leverage Abstrackr&#x2019;s relevance predictions in systematic and rapid reviews</article-title><source>BMC Med Res Methodol</source><year>2020</year><month>06</month><day>3</day><volume>20</volume><issue>1</issue><fpage>139</fpage><pub-id pub-id-type="doi">10.1186/s12874-020-01031-w</pub-id><pub-id pub-id-type="medline">32493228</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamel</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kelly</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Thavorn</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rice</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Wells</surname><given-names>GA</given-names> </name><name name-style="western"><surname>Hutton</surname><given-names>B</given-names> </name></person-group><article-title>An evaluation of DistillerSR&#x2019;s machine learning-based prioritization tool for title/abstract screening - impact on reviewer-relevant outcomes</article-title><source>BMC Med Res Methodol</source><year>2020</year><month>10</month><day>15</day><volume>20</volume><issue>1</issue><fpage>256</fpage><pub-id pub-id-type="doi">10.1186/s12874-020-01129-1</pub-id><pub-id pub-id-type="medline">33059590</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferdinands</surname><given-names>G</given-names> </name><name name-style="western"><surname>Schram</surname><given-names>R</given-names> </name><name name-style="western"><surname>de Bruin</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Performance of active learning models for screening prioritization in systematic reviews: a simulation study into the average time to discover relevant records</article-title><source>Syst Rev</source><year>2023</year><month>06</month><day>20</day><volume>12</volume><issue>1</issue><fpage>100</fpage><pub-id pub-id-type="doi">10.1186/s13643-023-02257-7</pub-id><pub-id pub-id-type="medline">37340494</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gilardi</surname><given-names>F</given-names> </name><name name-style="western"><surname>Alizadeh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kubli</surname><given-names>M</given-names> </name></person-group><article-title>ChatGPT outperforms crowd workers for text-annotation tasks</article-title><source>Proc Natl Acad Sci U S A</source><year>2023</year><month>07</month><day>25</day><volume>120</volume><issue>30</issue><fpage>e2305016120</fpage><pub-id pub-id-type="doi">10.1073/pnas.2305016120</pub-id><pub-id pub-id-type="medline">37463210</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Loukas</surname><given-names>L</given-names> </name><name name-style="western"><surname>Stogiannidis</surname><given-names>I</given-names> </name><name name-style="western"><surname>Malakasiotis</surname><given-names>P</given-names> </name><name name-style="western"><surname>Vassos</surname><given-names>S</given-names> </name></person-group><article-title>Breaking the bank with ChatGPT: few-shot text classification for finance</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 8, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2308.14634</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>M</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Park</surname><given-names>YJ</given-names> </name><name name-style="western"><surname>Paget</surname><given-names>M</given-names> </name><name name-style="western"><surname>Naugler</surname><given-names>C</given-names> </name></person-group><article-title>Automated paper screening for clinical reviews using large language models: data analysis study</article-title><source>J Med Internet Res</source><year>2024</year><month>01</month><day>12</day><volume>26</volume><fpage>e48996</fpage><pub-id pub-id-type="doi">10.2196/48996</pub-id><pub-id pub-id-type="medline">38214966</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zink</surname><given-names>J</given-names> </name><name name-style="western"><surname>Putora</surname><given-names>PM</given-names> </name><name name-style="western"><surname>Hastings</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cihoric</surname><given-names>N</given-names> </name></person-group><article-title>Title and abstract screening for literature reviews using large language models: an exploratory study in the biomedical domain</article-title><source>Syst Rev</source><year>2024</year><month>06</month><day>15</day><volume>13</volume><issue>1</issue><fpage>158</fpage><pub-id pub-id-type="doi">10.1186/s13643-024-02575-4</pub-id><pub-id pub-id-type="medline">38879534</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tran</surname><given-names>VT</given-names> </name><name name-style="western"><surname>Gartlehner</surname><given-names>G</given-names> </name><name name-style="western"><surname>Yaacoub</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Sensitivity and specificity of using GPT-3.5 Turbo models for title and abstract screening in systematic reviews and meta-analyses</article-title><source>Ann Intern Med</source><year>2024</year><month>06</month><volume>177</volume><issue>6</issue><fpage>791</fpage><lpage>799</lpage><pub-id pub-id-type="doi">10.7326/M23-3389</pub-id><pub-id pub-id-type="medline">38768452</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Matsui</surname><given-names>K</given-names> </name><name name-style="western"><surname>Utsumi</surname><given-names>T</given-names> </name><name name-style="western"><surname>Aoki</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Maruki</surname><given-names>T</given-names> </name><name name-style="western"><surname>Takeshima</surname><given-names>M</given-names> </name><name name-style="western"><surname>Takaesu</surname><given-names>Y</given-names> </name></person-group><article-title>Human-comparable sensitivity of large language models in identifying eligible studies through title and abstract screening: 3-layer strategy using GPT-3.5 and GPT-4 for systematic reviews</article-title><source>J Med Internet Res</source><year>2024</year><month>08</month><day>16</day><volume>26</volume><fpage>e52758</fpage><pub-id pub-id-type="doi">10.2196/52758</pub-id><pub-id pub-id-type="medline">39151163</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>X</given-names> </name></person-group><article-title>Evaluating the effectiveness of large language models in abstract screening: a comparative analysis</article-title><source>Syst Rev</source><year>2024</year><month>08</month><day>21</day><volume>13</volume><issue>1</issue><fpage>219</fpage><pub-id pub-id-type="doi">10.1186/s13643-024-02609-x</pub-id><pub-id pub-id-type="medline">39169386</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chai</surname><given-names>KEK</given-names> </name><name name-style="western"><surname>Lines</surname><given-names>RLJ</given-names> </name><name name-style="western"><surname>Gucciardi</surname><given-names>DF</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>L</given-names> </name></person-group><article-title>Research screener: a machine learning tool to semi-automate abstract screening for systematic reviews</article-title><source>Syst Rev</source><year>2021</year><month>04</month><day>1</day><volume>10</volume><issue>1</issue><fpage>93</fpage><pub-id pub-id-type="doi">10.1186/s13643-021-01635-3</pub-id><pub-id pub-id-type="medline">33795003</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Papaioannou</surname><given-names>D</given-names> </name><name name-style="western"><surname>Brazier</surname><given-names>J</given-names> </name><name name-style="western"><surname>Paisley</surname><given-names>S</given-names> </name></person-group><article-title>Systematic searching and selection of health state utility values from the literature</article-title><source>Value Health</source><year>2013</year><month>06</month><volume>16</volume><issue>4</issue><fpage>686</fpage><lpage>695</lpage><pub-id pub-id-type="doi">10.1016/j.jval.2013.02.017</pub-id><pub-id pub-id-type="medline">23796303</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arber</surname><given-names>M</given-names> </name><name name-style="western"><surname>Garcia</surname><given-names>S</given-names> </name><name name-style="western"><surname>Veale</surname><given-names>T</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shaw</surname><given-names>A</given-names> </name><name name-style="western"><surname>Glanville</surname><given-names>JM</given-names> </name></person-group><article-title>Performance of ovid medline search filters to identify health state utility studies</article-title><source>Int J Technol Assess Health Care</source><year>2017</year><month>01</month><volume>33</volume><issue>4</issue><fpage>472</fpage><lpage>480</lpage><pub-id pub-id-type="doi">10.1017/S0266462317000897</pub-id><pub-id pub-id-type="medline">29065942</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>Group TE</collab></person-group><article-title>EuroQol - a new facility for the measurement of health-related quality of life</article-title><source>Health Policy</source><year>1990</year><month>12</month><volume>16</volume><issue>3</issue><fpage>199</fpage><lpage>208</lpage><pub-id pub-id-type="doi">10.1016/0168-8510(90)90421-9</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rencz</surname><given-names>F</given-names> </name><name name-style="western"><surname>Brodszky</surname><given-names>V</given-names> </name><name name-style="western"><surname>Gul&#x00E1;csi</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Parallel valuation of the EQ-5D-3L and EQ-5D-5L by time trade-off in Hungary</article-title><source>Value Health</source><year>2020</year><month>09</month><volume>23</volume><issue>9</issue><fpage>1235</fpage><lpage>1245</lpage><pub-id pub-id-type="doi">10.1016/j.jval.2020.03.019</pub-id><pub-id pub-id-type="medline">32940242</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Herdman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gudex</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lloyd</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Development and preliminary testing of the new five-level version of EQ-5D (EQ-5D-5L)</article-title><source>Qual Life Res</source><year>2011</year><month>12</month><volume>20</volume><issue>10</issue><fpage>1727</fpage><lpage>1736</lpage><pub-id pub-id-type="doi">10.1007/s11136-011-9903-x</pub-id><pub-id pub-id-type="medline">21479777</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Golicki</surname><given-names>D</given-names> </name><name name-style="western"><surname>M&#x0142;y&#x0144;czak</surname><given-names>K</given-names> </name></person-group><article-title>Measurement properties of the EQ-5D-Y: a systematic review</article-title><source>Value Health</source><year>2022</year><month>11</month><volume>25</volume><issue>11</issue><fpage>1910</fpage><lpage>1921</lpage><pub-id pub-id-type="doi">10.1016/j.jval.2022.05.013</pub-id><pub-id pub-id-type="medline">35752534</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kingma</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Ba</surname><given-names>J</given-names> </name></person-group><article-title>Adam: a method for stochastic optimization</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 30, 2017</comment><pub-id pub-id-type="doi">10.48550/arXiv.1412.6980</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de la Torre-L&#x00F3;pez</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ram&#x00ED;rez</surname><given-names>A</given-names> </name><name name-style="western"><surname>Romero</surname><given-names>JR</given-names> </name></person-group><article-title>Artificial intelligence to automate the systematic review of scientific literature</article-title><source>Computing</source><year>2023</year><month>10</month><volume>105</volume><issue>10</issue><fpage>2171</fpage><lpage>2194</lpage><pub-id pub-id-type="doi">10.1007/s00607-023-01181-x</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Sundaram</surname><given-names>G</given-names> </name><name name-style="western"><surname>Berleant</surname><given-names>D</given-names> </name></person-group><article-title>Automating systematic literature reviews with natural language processing and text mining: a systematic literature review</article-title><source>International Congress on Information and Communication Technology</source><year>2023</year><publisher-name>Springer</publisher-name><fpage>73</fpage><lpage>92</lpage><pub-id pub-id-type="doi">10.1007/978-981-99-3243-6_7</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hasny</surname><given-names>M</given-names> </name><name name-style="western"><surname>Vasile</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Gianni</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bannach-Brown</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nasser</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mackay</surname><given-names>M</given-names> </name><etal/></person-group><article-title>BERT for complex systematic review screening to support the future of medical research</article-title><conf-name>International conference on artificial intelligence in medicine</conf-name><conf-date>Jun 12-15, 2023</conf-date><conf-loc>Portoro&#x017E;, Slovenia</conf-loc><fpage>173</fpage><lpage>182</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-34344-5_21</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moher</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hopewell</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schulz</surname><given-names>KF</given-names> </name><etal/></person-group><article-title>CONSORT 2010 explanation and elaboration: updated guidelines for reporting parallel group randomised trials</article-title><source>BMJ</source><year>2010</year><month>03</month><day>23</day><volume>340</volume><fpage>c869</fpage><pub-id pub-id-type="doi">10.1136/bmj.c869</pub-id><pub-id pub-id-type="medline">20332511</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hopewell</surname><given-names>S</given-names> </name><name name-style="western"><surname>Clarke</surname><given-names>M</given-names> </name><name name-style="western"><surname>Moher</surname><given-names>D</given-names> </name><etal/></person-group><article-title>CONSORT for reporting randomised trials in journal and conference abstracts</article-title><source>The Lancet</source><year>2008</year><month>01</month><volume>371</volume><issue>9609</issue><fpage>281</fpage><lpage>283</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(07)61835-2</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Simera</surname><given-names>I</given-names> </name><name name-style="western"><surname>Moher</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hirst</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hoey</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schulz</surname><given-names>KF</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>DG</given-names> </name></person-group><article-title>Transparent and accurate reporting increases reliability, utility, and impact of your research: reporting guidelines and the EQUATOR Network</article-title><source>BMC Med</source><year>2010</year><month>04</month><day>26</day><volume>8</volume><fpage>1</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1186/1741-7015-8-24</pub-id><pub-id pub-id-type="medline">20420659</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="web"><article-title>Towards automating the selection of articles reporting EQ-5D data for systematic literature reviews using large language models</article-title><source>GitHub</source><access-date>2026-08-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/kerteszg/eq5d_report_predictor">https://github.com/kerteszg/eq5d_report_predictor</ext-link></comment></nlm-citation></ref></ref-list></back></article>