<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e84989</article-id><article-id pub-id-type="doi">10.2196/84989</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Performance of Two AI Approaches in ASReview Compared With Manual Screening for Dementia Care Literature Screening: Comparative Analysis</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Steijger</surname><given-names>Dirk</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Thissen</surname><given-names>Stella</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Aarts</surname><given-names>Sil</given-names></name><degrees>Dr</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Verbeek</surname><given-names>Hilde</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>de Vugt</surname><given-names>Marjolein E</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Christie</surname><given-names>Hannah</given-names></name><degrees>Dr</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Psychiatry and Neuropsychology, Mental Health and Neuroscience Research Institute, Faculty of Health, Medicine and Life Sciences, Maastricht University</institution><addr-line>Dr Tanslaan, 12</addr-line><addr-line>Maastricht</addr-line><country>The Netherlands</country></aff><aff id="aff2"><institution>Department of Health Service Research, CAPHRI Care and Public Health Research Institute, Faculty of Health Medicine and Life Sciences, Maastricht University</institution><addr-line>Maastricht</addr-line><country>The Netherlands</country></aff><aff id="aff3"><institution>The Living Lab in Ageing &#x0026; Long-Term Care</institution><addr-line>Maastricht</addr-line><country>The Netherlands</country></aff><aff id="aff4"><institution>Department of Primary and Community care, Research Institute for Medical Innovation, Radboud university medical center</institution><addr-line>Nijmegen</addr-line><country>The Netherlands</country></aff><aff id="aff5"><institution>School of Population Health, Royal College of Surgeons</institution><addr-line>Dublin</addr-line><country>Ireland</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Scheper</surname><given-names>Mark</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yu</surname><given-names>Yunguo</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Dirk Steijger, MSc, Department of Psychiatry and Neuropsychology, Mental Health and Neuroscience Research Institute, Faculty of Health, Medicine and Life Sciences, Maastricht University, Dr Tanslaan, 12, Maastricht, 6229ET, The Netherlands, 31 612365497; <email>dirk.steijger@maastrichtuniversity.nl</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>20</day><month>8</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e84989</elocation-id><history><date date-type="received"><day>29</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>09</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>10</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Dirk Steijger, Stella Thissen, Sil Aarts, Hilde Verbeek, Marjolein de Vugt, Hannah Christie. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 20.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e84989"/><abstract><sec><title>Background</title><p>Literature reviews rely on rigorous title and abstract screening by researchers, which is time-consuming. AI-assisted literature screening tools have been proposed to improve efficiency by prioritizing titles and abstracts with the highest likelihood of meeting the inclusion criteria, thereby reducing the need to screen all records.</p></sec><sec><title>Objective</title><p>This study aims to evaluate the performance of two AI-assisted screening approaches in ASReview (version 1.3; Department of Methodology and Statistics, Utrecht University) compared with manual title and abstract screening in a previously completed and published scoping review on how AI can support the quality of life in people with dementia.</p></sec><sec sec-type="methods"><title>Methods</title><p>This study used a dataset of 4690 titles and abstracts from a published scoping review. The manual screening decisions of the scoping review served as the reference standard. Both ASReview approaches were applied by the same author who conducted the majority of the original manual title and abstract screening. Approach A used a simpler model with minimal prior input, whereas approach B used a more advanced model with a larger training set. Both ASReview approaches applied predefined stopping rules: (1) more than 10% of the dataset to be screened; and (2) 50 consecutive irrelevant titles and abstracts. Performance was evaluated in terms of sensitivity, specificity, precision, accuracy, and screening time. 95% CIs were calculated for sensitivity, specificity, precision, and accuracy. Agreement between manual and ASReview approaches was assessed using the Cohen &#x03BA;, and differences in how manual and both ASReview approaches classified titles and abstracts were examined using the McNemar test. Performance and agreement were calculated at two levels: (1) after title and abstract screening and (2) after full-text inclusion.</p></sec><sec sec-type="results"><title>Results</title><p>Manual screening identified 283 titles and abstracts for full-text review and resulted in 30 final included studies, requiring 19 hours. Of the 4690 titles and abstracts, approach A screened 830 (17.7%) in 4.3 hours and retrieved 16 of the 30 (sensitivity 0.53, 95% CI 0.36&#x2010;0.70) final included studies, whereas approach B screened 798 (17.0%) in 5.5 hours and retrieved 21 of the 30 (sensitivity 0.70, 95% CI 0.52&#x2010;0.83) final included studies. Although both ASReview approaches showed high specificity and accuracy, these metrics should be interpreted cautiously because the dataset was highly imbalanced and contained relatively few relevant titles and abstracts. McNemar tests showed significant directional imbalance (<italic>P</italic>&#x003C;.001): ASReview missed more manually selected titles and abstracts at level 1, whereas at level 2, ASReview more often labeled titles and abstracts not included in the final review as relevant.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>ASReview can support workload reduction in title and abstract screening, but the evaluated ASReview approaches did not retrieve all final included studies from the original dementia care scoping review. These findings suggest that the evaluated ASReview configurations may be insufficient for reviews in which near-complete retrieval of relevant evidence is required.</p></sec></abstract><kwd-group><kwd>ASReview</kwd><kwd>dementia</kwd><kwd>literature screening</kwd><kwd>AI</kwd><kwd>information retrieval</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>In health research, evidence synthesis is used to integrate and summarize existing literature and to inform future research, guidelines, policy, and decision-making [<xref ref-type="bibr" rid="ref1">1</xref>]. Title and abstract screening is a tedious but vital step in this process [<xref ref-type="bibr" rid="ref2">2</xref>]. The purpose of title and abstract screening is to identify those that are potentially eligible for full-text assessment and to exclude those that clearly do not meet the eligibility criteria [<xref ref-type="bibr" rid="ref2">2</xref>]. Although title and abstract screening is only 1 component of the evidence synthesis process, it is a key step in determining which records proceed to full-text review. Overlooking relevant titles and abstracts during screening can introduce bias and affect the completeness of the evidence base [<xref ref-type="bibr" rid="ref1">1</xref>]. Because the title and abstract screening phase often involves a large number of titles and abstracts, it is time-consuming and cognitively demanding [<xref ref-type="bibr" rid="ref3">3</xref>]. Therefore, efficient and innovative approaches to support the title and abstract screening process are needed.</p><p>To streamline the title and abstract screening process, various AI-assisted screening tools have emerged [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. These AI-assisted screening tools may use techniques such as text mining, machine learning, natural language processing, and deep learning to identify and actively learn patterns in titles and abstracts and prioritize titles and abstracts that are more likely to be relevant. AI-assisted screening can be implemented in specialized screening tools, such as ASReview (version 1.3; Department of Methodology and Statistics, Utrecht University) [<xref ref-type="bibr" rid="ref6">6</xref>], and Abstrackr [<xref ref-type="bibr" rid="ref9">9</xref>] or as AI-supported functions within broader review management platforms, such as DistillerSR [<xref ref-type="bibr" rid="ref10">10</xref>], Covidence [<xref ref-type="bibr" rid="ref11">11</xref>], and Rayyan [<xref ref-type="bibr" rid="ref12">12</xref>]. These AI-assisted screening tools are intended to support reviewer decision-making by prioritizing titles and abstracts based on predicted relevance [<xref ref-type="bibr" rid="ref5">5</xref>]. By identifying likely relevant studies early, these AI-assisted screening tools could potentially allow reviewers to screen only a subset of the dataset, thereby reducing workload.</p><p>Previous studies evaluating AI-assisted screening tools in health research have shown that these tools can potentially reduce screening time while maintaining the ability to identify relevant titles and abstracts [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref13">13</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. Focusing specifically on ASReview, a widely used open-source AI-assisted screening tool [<xref ref-type="bibr" rid="ref18">18</xref>], previous evaluations have reported promising performance in different review contexts [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. The performance of AI-assisted screening tools depends on the effectiveness of the underlying AI techniques and algorithms, the quality and quantity of data used for training, and the degree of human involvement [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. Therefore, further evaluations are needed to understand how ASReview performs in title and abstract screening when different model configurations are applied across different review contexts.</p><p>Given this dependence on context, the intersection of dementia care, quality of life, and AI provides a relevant context for further evaluation. This intersection provides a useful starting point for examining how ASReview performs in a review area that combines dementia care research with technology-oriented literature. To our knowledge, ASReview has not yet been evaluated in such a dementia care literature review context. Therefore, this study aimed to evaluate the performance of two AI-assisted screening approaches within ASReview compared with manual title and abstract screening in a previously completed and published scoping review on how AI can support quality of life in people with dementia [<xref ref-type="bibr" rid="ref26">26</xref>]. Specifically, this study examined the extent to which both ASReview approaches reduced screening workload and identified the same titles and abstracts as included via manual screening in the scoping review. By comparing title and abstract screening in ASReview with a manual screening process, this study adds empirical evidence on the practical use of AI-assisted screening in a real-world dementia care review context.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This study was designed as a comparative methodological evaluation of two AI-assisted screening approaches within ASReview compared with the manual screening approach used in a previously published scoping review on how AI can support quality of life in people with dementia [<xref ref-type="bibr" rid="ref26">26</xref>]. Reporting of the present study was guided by the PRISMA-trAIce (Preferred Reporting Items for Systematic Reviews and Meta-Analyses&#x2013;Transparent Reporting of AI-Assisted Evidence Synthesis) guidelines for the transparent reporting of AI-assisted evidence synthesis, where applicable [<xref ref-type="bibr" rid="ref27">27</xref>].</p></sec><sec id="s2-2"><title>Unit of Analysis, Data Source, and Study Size</title><p>The unit of analysis was the individual title and abstract record. The data were derived from the title and abstract dataset of the original scoping review, including both the manual screening decisions and the final full-text inclusion decisions. Manual screening and inclusion decisions from the original review served as the reference standard against which the two ASReview approaches were evaluated. The study size corresponded to that of the full screening dataset from the original scoping review. After deduplication and preparation for screening, the dataset consisted of 4690 titles and abstracts, of which 283 were selected for full-text review and 30 were included in the final review.</p></sec><sec id="s2-3"><title>Compared Approaches</title><sec id="s2-3-1"><title>Manual Approach</title><p>The manual screening approach was performed as part of a previously published scoping review [<xref ref-type="bibr" rid="ref26">26</xref>]. In brief, the original review searched PubMed, Scopus, the Association for Computing Machinery Digital Library, and Google Scholar for studies published between 2010 and January 2024 on AI-based approaches that could support the quality of life of people with dementia. Eligible studies involved people with at least suspected dementia, evaluated an AI-based approach to the daily living of people with dementia, and reported original empirical research. Studies were excluded if they focused on diagnostic or acute care applications, if they were not available in English or Dutch, or if the full text was unavailable.</p><p>After deduplication, records were imported into Rayyan and screened, without the use of the AI function in Rayyan, against the eligibility criteria. Initially, 2 reviewers (DS with 5 years of research experience and the other reviewer HC with 10 years of research experience) independently screened 10% of the titles and abstracts from the scientific databases. As the interrater agreement exceeded the predefined threshold (80%), 1 reviewer (DS) continued screening the remaining titles and abstracts. The same procedure was applied to full-text screening. For the Google Scholar gray literature search, 1 reviewer (DS) assessed the first 300 results. However, no Google Scholar records were included in the final review or in the dataset used for the present ASReview evaluation. A detailed description of the search strategy, eligibility criteria, and screening process is reported in the original scoping review [<xref ref-type="bibr" rid="ref26">26</xref>]. The resulting manual title and abstract screening decisions and final full-text inclusion decisions were used as the reference standards for the present methodological evaluation.</p></sec><sec id="s2-3-2"><title>ASReview</title><p>ASReview is an open-source AI-assisted screening tool that uses active learning to prioritize titles and abstracts according to their predicted relevance [<xref ref-type="bibr" rid="ref6">6</xref>]. Before the title and abstract screening starts, ASReview requires prior knowledge consisting of titles and abstracts labeled as relevant and irrelevant by the reviewer. Based on this prior knowledge, the tool ranks the remaining titles and abstracts from highest to lowest predicted relevance. The reviewer then screens the highest-ranked title and abstract and labels it as relevant or irrelevant according to the eligibility criteria. After each reviewer&#x2019;s decision, ASReview updates the ranking and presents the next title and abstract with the highest predicted relevance. This researcher-in-the-loop process continues until a predefined stopping rule is reached. ASReview allows reviewers to preselect different model configurations, including feature extraction methods (ie, Doc2Vec, embedding inverse document frequency, term frequency&#x2013;inverse document frequency [TF-IDF], and Sentence-Bidirectional Encoder Representations from Transformers [sBERT]) and classifiers (Na&#x00EF;ve Bayes, support vector machines, deep neural network, logistic regression, long short-term memory, and random forest) [<xref ref-type="bibr" rid="ref6">6</xref>]. In the present study, ASReview was applied to the same dataset and eligibility criteria as the manual screening approach. This allowed a direct comparison between the original manual screening decisions and the screening decisions generated through two AI-assisted ASReview configurations. The two ASReview configurations were based on recommendations from ASReview&#x2019;s model selection guide and the SAFE (Screen, Apply, Find, Evaluate) procedure [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]: a default configuration and a custom configuration.</p></sec><sec id="s2-3-3"><title>Approach A: Default Configuration</title><p>Approach A was included as a baseline AI-assisted screening approach because it reflects ASReview&#x2019;s default configuration and requires minimal prior-knowledge input from the reviewer [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. It used term TF-IDF for feature extraction, combined with a Na&#x00EF;ve Bayes classifier. The query strategy was set to maximum, and the balance strategy was set to dynamic resampling (double). TF-IDF represents titles and abstracts based on word frequencies, while Na&#x00EF;ve Bayes uses these patterns to estimate the likelihood of relevance. Four titles and abstracts, consisting of 1 relevant and 3 irrelevant examples, were used as prior knowledge. The prior-knowledge titles and abstracts were included in the analysis as classified titles and abstracts by ASReview.</p></sec><sec id="s2-3-4"><title>Approach B: Custom Configuration</title><p>Approach B was included to evaluate whether a more semantically rich model configuration, combined with a larger prior knowledge set, would improve the retrieval of relevant titles and abstracts compared with the default configuration. Approach B used sBERT for feature extraction, combined with a fully connected neural network classifier. In ASReview version 1.3, sBERT-based feature extraction required an extension package. The exact extension package name and version used in the present analysis were not retained. The query strategy was set to maximum, and the balance strategy used dynamic resampling (double). sBERT represents titles and abstracts as dense semantic vectors that capture contextual meaning beyond individual word frequencies, while the neural network classifier uses these representations to estimate the likelihood of relevance. Prior knowledge consisted of 47 titles and abstracts, corresponding to approximately 1% of the original dataset. This set included 2 relevant titles and abstracts randomly selected from the 30 final included titles and abstracts and 45 irrelevant titles and abstracts randomly selected from the titles and abstracts that were excluded for full-text screening in the original review. This ensured that the prior knowledge set contained both relevant and irrelevant titles and abstracts, consistent with the SAFE procedure&#x2019;s recommendation to start with a labeled training set of approximately 1% of the dataset containing at least 1 relevant and 1 irrelevant title and abstract [<xref ref-type="bibr" rid="ref25">25</xref>]. Prior knowledge titles and abstracts were included in the analysis as classified titles and abstracts by ASReview.</p></sec></sec><sec id="s2-4"><title>Screening Procedure</title><p>The screening procedure and stopping rule were informed by selected elements of the SAFE procedure [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. The SAFE procedure provides practical guidance for active learning&#x2013;based screening, including the use of prior knowledge and stopping heuristics that combine a minimum screening proportion with a run of consecutive irrelevant titles and abstracts. Because the present study was a retrospective methodological evaluation rather than a live review selection process, the full SAFE procedure was not implemented. Instead, selected SAFE elements were used to define the prior-knowledge strategy and stopping rule for the ASReview screening approaches.</p><p>For the present study, 2 separate ASReview projects were created in the ASReview environment, 1 for approach A and 1 for approach B. For each project, the complete title and abstract dataset from the original scoping review were exported from Rayyan and uploaded to ASReview. Approach A and approach B were conducted as independent comparative screening projects. The screening decisions from approach A were not used as training data for approach B.</p><p>To ensure consistency in the application of the eligibility criteria, the same reviewer who conducted the majority of the title and abstract screening in the original scoping review (DS) also screened the titles and abstracts presented by ASReview for each approach. The same eligibility criteria as in the original scoping review were applied. Titles and abstracts were labeled as relevant or irrelevant independently for approach A and approach B.</p><p>Screening continued until the predefined stopping rule was met. This stopping rule required that 2 conditions were fulfilled: first, at least 10% of the complete dataset had to be screened; second, after this minimum threshold had been reached, screening continued until 50 consecutive titles and abstracts had been labeled as irrelevant by the reviewer. The threshold of 50 consecutive irrelevant titles and abstracts was selected as a pragmatic stopping criterion. The authors acknowledge that more conservative thresholds, such as 100 consecutive irrelevant titles and abstracts, may increase sensitivity by extending the screening process. The same stopping rule was applied to both ASReview approaches. Because each ASReview configuration generated a different ranking of titles and abstracts, the stopping rule was reached after a different number of screened titles and abstracts in each approach. After stopping, the titles and abstracts labeled as relevant by each ASReview approach were compared with the manual title and abstract screening decisions and final inclusion decisions from the original scoping review.</p></sec><sec id="s2-5"><title>Performance Outcomes</title><p>To assess performance, both ASReview approaches were compared with manual screening using sensitivity, specificity, precision, accuracy, screening time, and agreement. Sensitivity was used to assess the proportion of manually selected titles and abstracts that were also identified by each ASReview approach. Specificity was used to assess the proportion of manually excluded titles and abstracts that were also excluded by ASReview. Precision was used to assess the proportion of titles and abstracts labeled as relevant by each ASReview approach that were also relevant according to the manual approach. Accuracy was used to assess the overall proportion of titles and abstracts that were correctly classified by each ASReview approach, including both records considered relevant and records considered irrelevant according to the manual approach. Agreement between each ASReview approach and the manual approach was measured using the Cohen &#x03BA; and observed agreement (<italic>P</italic><sub>o</sub>) [<xref ref-type="bibr" rid="ref31">31</xref>], and the McNemar test was used to assess directional imbalance in discordant classifications [<xref ref-type="bibr" rid="ref32">32</xref>]. Total screening time was estimated using the time-tracking information displayed within Rayyan for the manual approach and within ASReview for both AI-assisted approaches. Screening time did not include full-text screening or data extraction. False negatives were examined descriptively to explore potential screening-related selection bias by assessing whether missed titles and abstracts shared common characteristics, such as dementia care context, AI methods, or failure to report AI performance. The evaluation level at which each metric was calculated is described in the Data Analysis section.</p></sec><sec id="s2-6"><title>Statistical Analysis</title><p>Performance was assessed at 2 levels. At level 1, manual title and abstract screening decisions were used as the reference standard. Thus, titles and abstracts selected for full-text review during manual screening were considered relevant, whereas titles and abstracts excluded during manual title and abstract screening were considered irrelevant. At this level, sensitivity, Cohen &#x03BA;, observed agreement, and the McNemar test were calculated.</p><p>At level 2, the final full-text inclusion decisions from the original scoping review were used as the reference standard: titles and abstracts included in the final review were considered relevant, whereas all other records were considered irrelevant. At this level, sensitivity, specificity, precision, accuracy, Cohen &#x03BA;, observed agreement, and the McNemar test were calculated. This approach aligns with prior research, suggesting that although ASReview retrieves fewer abstracts, a higher proportion may be included in the final inclusion set [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. For binary classification metrics, 95% CIs were calculated using Wilson score intervals [<xref ref-type="bibr" rid="ref34">34</xref>]. At both levels, prior knowledge titles and abstracts were included in the statistical analysis of the performance metrics.</p><p>Screening time referred only to the title and abstract screening process and did not include full-text screening (during the manual approach). For all approaches, total screening time was recorded in minutes at the end of the screening process.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p><xref ref-type="fig" rid="figure1">Figure 1</xref> summarizes the flow of titles and abstracts through the manual screening approach and the two ASReview approaches. The manual screening approach screened a highly imbalanced dataset of 4690 abstracts, identified 283 abstracts for full-text review, and included 30 final studies, requiring 19 hours. Approach A achieved 50 consecutive irrelevant abstracts and thus stopped screening at 830 (17.7%) abstracts in 4.3 hours, with a sensitivity of 0.16 at level 1 and 0.53 at level 2 (16 abstracts), specificity of 0.99, precision of 0.21, and accuracy of 0.98 at level 2. Approach B achieved 50 consecutive irrelevant abstracts and thus stopped screening at 798 (17.0%) abstracts in 5.5 hours, achieving higher sensitivity at both level 1: 0.23 and level 2: 0.70 (21 abstracts), specificity of 0.99, precision of 0.24, and accuracy of 0.98 at level 2. Both ASReview approaches reduced screening time compared with manual screening. At the final full-text inclusion decisions from the original scoping review stage, sensitivity was higher for approach B than for approach A (0.70 vs 0.53; an absolute difference of 0.17). The performance metrics are shown in <xref ref-type="table" rid="table1">Table 1</xref>. The 95% CIs indicate uncertainty around the sensitivity and precision estimates, particularly at level 2. For approach B, level 2 sensitivity was 0.70 (95% CI 0.52&#x2010;0.83) and level 2 precision was 0.24 (95% CI 0.16&#x2010;0.34).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flow diagram of titles and abstracts through two AI-assisted literature screening approaches within ASReview compared with manual title and abstract screening from a dementia care scoping review. The figure summarizes how the dataset from the original dementia care scoping review was screened manually and with two AI-assisted approaches within ASReview. After deduplication and preparation, the dataset consisted of 4690 titles and abstracts. Manual screening assessed all 4690 titles and abstracts, selected 283 for full-text review, and resulted in 30 final included studies. Approach A screened 830 titles and abstracts until the predefined stopping rule was reached, labeled 78 titles and abstracts as relevant, and retrieved 16 of the 30 final included studies. Approach B screened 798 titles and abstracts until the predefined stopping rule was reached, labeled 87 titles and abstracts as relevant, and retrieved 21 of the 30 final included studies.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e84989_fig01.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Performance of two AI-assisted literature screening approaches within ASReview compared with manual title and abstract screening from a dementia care scoping review.<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Approach</td><td align="left" valign="bottom">Sensitivity (level 1, 95% CI)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="bottom">Sensitivity (level 2, 95% CI)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="bottom">Specificity (level 2, 95% CI)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="bottom">Precision (level 2, 95% CI)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="bottom">Accuracy (level 2, 95% CI)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="bottom">Screened titles and abstract, n (%)</td><td align="left" valign="bottom">Screening time (h)<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Manual<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">4690 (100)</td><td align="left" valign="top">19</td></tr><tr><td align="left" valign="top">A</td><td align="left" valign="top">0.16 (0.12&#x2010;0.21)</td><td align="left" valign="top">0.53 (0.36&#x2010;0.70)</td><td align="left" valign="top">0.99 (0.98&#x2010;0.99)</td><td align="left" valign="top">0.21 (0.13&#x2010;0.31)</td><td align="left" valign="top">0.98 (0.98&#x2010;0.99)</td><td align="left" valign="top">830 (17.7)</td><td align="left" valign="top">4.3</td></tr><tr><td align="left" valign="top">B</td><td align="left" valign="top">0.23 (0.18&#x2010;0.28)</td><td align="left" valign="top">0.70 (0.52&#x2010;0.83)</td><td align="left" valign="top">0.99 (0.98&#x2010;0.99)</td><td align="left" valign="top">0.24 (0.16&#x2010;0.34)</td><td align="left" valign="top">0.98 (0.98&#x2010;0.99)</td><td align="left" valign="top">798 (17.0)</td><td align="left" valign="top">5.5</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>The dataset consisted of 4690 titles and abstracts from a previously published scoping review on how AI can support quality of life in people with dementia, covering studies published between 2010 and January 2024. </p></fn><fn id="table1fn2"><p><sup>b</sup>The CIs were calculated using Wilson score intervals.</p></fn><fn id="table1fn3"><p><sup>c</sup>Level 1 sensitivity was calculated using the 283 titles and abstracts selected for full-text review during manual screening as the reference standard. </p></fn><fn id="table1fn4"><p><sup>d</sup>Level 2 sensitivity, specificity, precision, and accuracy were calculated using the 30 titles and abstracts included in the final review as the reference standard. </p></fn><fn id="table1fn5"><p><sup>e</sup>Screening time refers only to title and abstract screening and does not include full-text screening.</p></fn><fn id="table1fn6"><p><sup>f</sup>Manual screening decisions from the original scoping review served as the reference standard; therefore, values of 1.00 for the manual approach are fixed by definition and should not be interpreted as evidence of perfect manual screening performance.</p></fn></table-wrap-foot></table-wrap><p>At level 1, agreement between manual screening and ASReview was moderate for both ASReview approaches (&#x03BA;=0.24, <italic>P</italic><sub>o</sub>=0.94 for approach A; &#x03BA;=0.24, <italic>P</italic><sub>o</sub>=0.95 for approach B). At level 2, Cohen &#x03BA; values increased slightly (&#x03BA;=0.28 for approach A; &#x03BA;=0.35 for approach B) with high observed agreement (<italic>P</italic><sub>o</sub>=0.98 for both). McNemar tests showed a significant but level-dependent directional imbalance (<italic>P</italic>&#x003C;.001). At level 1, ASReview missed more manually selected titles and abstracts than it additionally labeled as relevant: 237 vs 32 for approach A and 219 vs 23 for approach B. At level 2, the direction was reversed: ASReview labeled more titles and abstracts not included in the final review as relevant than it missed final included titles and abstracts, with 62 vs 14 for approach A and 66 vs 9 for approach B. Agreement statistics are presented in <xref ref-type="table" rid="table2">Table 2</xref>, with the corresponding 2 &#x00D7; 2 contingency tables provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Analysis of the false negatives revealed no dominant thematic pattern; false-negative abstracts covered both AI and dementia, with no specific topic being consistently missed.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Agreement and directional imbalance between two ASReview approaches and manual title and abstract screening from a dementia care scoping review.<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Level<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="bottom">Comparison<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="bottom">Cohen &#x03BA;<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="bottom">Observed agreement (<italic>P</italic><sub>o</sub>)<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="bottom">b<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="bottom">c<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="bottom">McNemar test (<italic>P</italic> value)<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">1</td><td align="left" valign="top">Manual vs A</td><td align="left" valign="top">0.24</td><td align="left" valign="top">0.94</td><td align="left" valign="top">237</td><td align="left" valign="top">32</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">1</td><td align="left" valign="top">Manual vs B</td><td align="left" valign="top">0.24</td><td align="left" valign="top">0.95</td><td align="left" valign="top">219</td><td align="left" valign="top">23</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">Manual vs A</td><td align="left" valign="top">0.28</td><td align="left" valign="top">0.98</td><td align="left" valign="top">14</td><td align="left" valign="top">62</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">Manual vs B</td><td align="left" valign="top">0.35</td><td align="left" valign="top">0.98</td><td align="left" valign="top">9</td><td align="left" valign="top">66</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>The dataset consisted of 4690 titles and abstracts from a previously published scoping review on AI-based approaches to support quality of life in people with dementia, covering studies published between 2010 and January 2024. </p></fn><fn id="table2fn2"><p><sup>b</sup>At level 1, titles and abstracts selected for full-text review during manual screening were considered relevant. At level 2, titles and abstracts included in the final scoping review were considered relevant. </p></fn><fn id="table2fn3"><p><sup>c</sup>Manual screening decisions from the original review served as the reference standard. </p></fn><fn id="table2fn4"><p><sup>d</sup>Agreement between each ASReview approach and manual screening was assessed using Cohen &#x03BA; and observed agreement (<italic>P</italic><sub>o</sub>).</p></fn><fn id="table2fn5"><p><sup>e</sup>The values b and c represent discordant classifications between manual screening and ASReview.</p></fn><fn id="table2fn6"><p><sup>f</sup>The McNemar test was used to assess directional imbalance in discordant classifications. </p></fn></table-wrap-foot></table-wrap></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study evaluated the performance of two AI-assisted screening approaches within ASReview compared with manual title and abstract screening in a previously completed dementia care scoping review. Both ASReview approaches substantially reduced screening time and the number of titles and abstracts that required screening, but neither approach retrieved all studies included in the final scoping review. Approach B (custom configuration), which used a semantically richer feature extraction method and a larger prior knowledge set, performed better than approach A (default configuration). However, approach B still missed several final included studies. These findings illustrate that ASReview can support workload reduction in the title and abstract screening process. However, the use of ASReview may be insufficient for reviews in which the aim is to identify and capture as much evidence as possible.</p><p>The finding that ASReview can reduce screening workload is in line with previous evaluations of active learning-based screening tools for title and abstract screening [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. However, there is no solid overview of how well active learning-based screening tools perform in the title and abstract screening process [<xref ref-type="bibr" rid="ref23">23</xref>], and previous evaluations show that performance varies across review topics, dataset size, model configurations, prior knowledge strategies, and stopping rules [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. This means that ASReview performance cannot be inferred from previous evaluations alone but should be assessed within the specific review context in which it is used. For active learning-based screening workflows, this has implications: a fixed minimum screening proportion, such as 10% or 50% of the total dataset, cannot by itself ensure full identification of relevant titles and abstracts [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. Sensitivity also depends on the model configuration and the composition of the prior knowledge set [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. All these implications may be particularly important in review topics with heterogeneous terminology, where relevant titles and abstracts may be described indirectly or inconsistently [<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>The high specificity and accuracy observed for both ASReview approaches should be interpreted in light of the highly imbalanced dataset. In title and abstract screening, irrelevant titles and abstracts typically form the large majority of the dataset [<xref ref-type="bibr" rid="ref39">39</xref>]. As a result, high specificity and accuracy can mainly reflect the correct exclusion of irrelevant titles and abstracts rather than the successful identification of relevant titles and abstracts [<xref ref-type="bibr" rid="ref40">40</xref>]. Therefore, sensitivity, false negatives, and the number of missed final included studies are particularly important for assessing whether an AI-assisted screening workflow is suitable for evidence synthesis.</p><p>The 70% (21/30) sensitivity observed for the best-performing ASReview approach represents a critical limitation for using the evaluated workflow as a screening strategy. Approach B retrieved 21 of the 30 final included studies from the original review, meaning that 9 studies would have been missed if this approach had been used as the primary screening approach during the original review. This level of sensitivity would be insufficient for reviews in which near-complete capturing of evidence is expected [<xref ref-type="bibr" rid="ref41">41</xref>]. By comparison, dual-reviewer screening of titles and abstracts misses around 2.5% of the relevant titles and abstracts [<xref ref-type="bibr" rid="ref42">42</xref>]. However, dual-reviewer screening is not always feasible because it requires substantial resources [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. This trade-off could be acceptable for rapid reviews, where reviewers are willing to give up some degree of certainty regarding the comprehensiveness of the included evidence [<xref ref-type="bibr" rid="ref42">42</xref>]. However, it remains unclear what level of reduced sensitivity is acceptable when AI-assisted screening tools are used. Further research is therefore needed to determine what level of reduced sensitivity is acceptable when using AI-assisted screening tools in different review contexts.</p><p>Agreement analysis highlighted the effect of class imbalance in title and abstract screening. Observed agreement between ASReview and manual screening was high; this was largely driven by shared exclusions of titles and abstracts. In contrast, Cohen &#x03BA; values ranged from 0.24 to 0.35, which indicate poor agreement beyond chance [<xref ref-type="bibr" rid="ref43">43</xref>]. This suggests that, despite high observed agreement, ASReview showed limited agreement with manual screening on which titles and abstracts should be selected as relevant. This is expected in title and abstract screening, where relevant titles and abstracts are typically a small proportion of the dataset, but the low kappa values should not be interpreted as acceptable agreement [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. The McNemar test provided complementary information by showing that the direction of discordance differed between evaluation levels [<xref ref-type="bibr" rid="ref45">45</xref>]. At level 1, ASReview missed more titles and abstracts selected for full-text review by manual screening than it additionally labeled as relevant. At level 2, the direction was reversed: ASReview labeled more titles and abstracts that were not included in the final review as relevant than it missed among the final included titles and abstracts. Together, these findings show that high observed agreement could mask practically important disagreement. Missed final included titles and abstracts may reduce the completeness of the evidence base, whereas level 2 false-positive titles and abstracts indicate that workload reduction was accompanied by limited precision among ASReview-selected titles and abstracts.</p><p>Current study findings should be interpreted in the context of a broader movement toward responsible use of AI in evidence synthesis. The RAISE (Responsible Use of AI in Evidence Synthesis) recommendations, developed by Cochrane, among others, provide a framework for ensuring the responsible use of AI across all roles within the evidence synthesis process [<xref ref-type="bibr" rid="ref46">46</xref>]. Similarly, PRISMA-trAIce frameworks provide a reporting framework to improve transparency when AI is used as a methodological tool in evidence synthesis [<xref ref-type="bibr" rid="ref27">27</xref>]. These developments may support broader adoption of AI-assisted screening tools in the title and abstract screening process but also increase the need for transparent reporting and validation of how such tools are used in specific review contexts [<xref ref-type="bibr" rid="ref37">37</xref>]. Retrospective validation studies may help identify context-specific best practices for using active learning during title and abstract screening [<xref ref-type="bibr" rid="ref47">47</xref>]. However, a lack of dataset availability can hinder reproducibility [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. The present study contributes to this evidence by evaluating ASReview&#x2019;s performance in the context of a dementia care scoping review and by adopting an open science approach, making the full dataset and reviewer decisions publicly available so that the current study is reproducible.</p></sec><sec id="s4-2"><title>Strengths and Limitations</title><p>To our knowledge, this is the first study to evaluate the performance of ASReview in the dementia care literature research. By making the full title and abstract dataset and all screening decisions available, this study also contributes to transparency, reproducibility, and future validation of AI-assisted screening in this review context. Another strength is that the same researcher conducted both the manual and AI-assisted screening, which helped ensure consistent adherence to the eligibility criteria across all screening approaches. However, the current results need to be viewed in light of some possible limitations. This study focused on a single scoping review, which limits generalizability to other review topics. Second, the manual screening and final inclusion decisions from the original scoping review were used as the reference standard for evaluating both ASReview approaches. Although the original review included an initial dual-screening validation procedure, the final screening decisions should not be interpreted as an error-free gold standard. Third, the ASReview screening approaches were conducted retrospectively by a reviewer who had been involved in the original scoping review title and abstract screening process. Therefore, the reviewer may have been more familiar with the review topic, eligibility criteria, and terminology during the ASReview screening. This familiarity may have influenced labeling decisions and may also have reduced the time needed to assess titles and abstracts compared with the original manual screening process, in which the reviewer still had to become familiar with the literature and screening criteria. Finally, this evaluation was conducted using ASReview version 1.3. The exact extension package name and version used to enable sBERT-based feature extraction in approach B were not retained. Because ASReview and its extensions are under active development, findings may differ with newer releases, plugin versions, alternative model configurations, or updated workflow options.</p></sec><sec id="s4-3"><title>Conclusions</title><p>This study evaluated two AI-assisted screening approaches within ASReview against manual title and abstract screening in a completed dementia care scoping review. Both ASReview approaches reduced screening time and the number of titles and abstracts requiring manual screening, but neither approach retrieved all titles and abstracts included in the final review. These findings suggest that AI-assisted screening tools, in particular ASReview, can support workload reduction in title and abstract screening but that both evaluated ASReview approaches may be insufficient for reviews in which near-complete capturing of evidence is required. Further research is needed to determine how AI-assisted screening tools perform in review contexts and to determine what level of reduced sensitivity is acceptable when using AI-assisted screening tools in different review contexts.</p></sec></sec></body><back><ack><p>During the preparation and revision of this manuscript, the corresponding author (DS) used ChatGPT (OpenAI) to support language editing and to assist with drafting responses to reviewer comments. The tool was not used for data analysis, interpretation of the results, generation of references, or autonomous scientific decision-making. All AI-assisted text was critically reviewed, edited, and approved by the authors. The authors take full responsibility for the content of this manuscript.</p></ack><notes><sec><title>Funding</title><p>This research was funded by the Dutch Research Council (NWO) through the QoLEAD (Quality of Life by use of Enabling AI in Dementia) project (project: KICH1.GZ02.20.008). Additional support from Alzheimer Nederland is gratefully acknowledged.</p></sec><sec><title>Data Availability</title><p>All data generated or analyzed during this study are included in the supplementary information files. The original title and abstract dataset from the published scoping review, including the manual screening decisions and final inclusion decisions, is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The screening decisions generated by both ASReview approaches, as well as the titles and abstracts used as prior knowledge for each ASReview approach, are also provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: DS, ST, SA, HV, MEdV, HC</p><p>Formal analysis: DS</p><p>Investigation: DS, ST</p><p>Methodology: DS, ST, SA, HV, MEdV, HC</p><p>Supervision: HV, MEdV, HC</p><p>Validation: SA</p><p>Visualization: DS, SA</p><p>Writing &#x2013; original draft: DS, ST</p><p>Writing &#x2013; review &#x0026; editing: DS, ST, SA, HV, MEdV, HC</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ACM</term><def><p>Association for Computing Machinery</p></def></def-item><def-item><term id="abb2">PRISMA-trAIce</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses&#x2013;Transparent Reporting of AI-Assisted Evidence Synthesis</p></def></def-item><def-item><term id="abb3">RAISE</term><def><p>Responsible Use of AI in Evidence Synthesis</p></def></def-item><def-item><term id="abb4">SAFE</term><def><p>Screen, Apply, Find, Evaluate</p></def></def-item><def-item><term id="abb5">SBERT</term><def><p>Sentence-Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb6">TF-IDF</term><def><p>Term Frequency&#x2013;Inverse Document Frequency</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Higgins</surname><given-names>JPT</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chandler</surname><given-names>J</given-names> </name><etal/></person-group><source>Cochrane Handbook for Systematic Reviews of Interventions</source><year>2019</year><access-date>2026-07-22</access-date><publisher-name>Wiley</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://dariososafoula.wordpress.com/wp-content/uploads/2017/01/cochrane-handbook-for-systematic-reviews-of-interventions-2019-1.pdf">https://dariososafoula.wordpress.com/wp-content/uploads/2017/01/cochrane-handbook-for-systematic-reviews-of-interventions-2019-1.pdf</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Polanin</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Pigott</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Espelage</surname><given-names>DL</given-names> </name><name name-style="western"><surname>Grotpeter</surname><given-names>JK</given-names> </name></person-group><article-title>Best practice guidelines for abstract screening large&#x2010;evidence systematic reviews and meta&#x2010;analyses</article-title><source>Res Synth Methods</source><year>2019</year><month>09</month><volume>10</volume><issue>3</issue><fpage>330</fpage><lpage>342</lpage><pub-id pub-id-type="doi">10.1002/jrsm.1354</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nussbaumer-Streit</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ellen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Klerings</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Resource use during systematic review production varies widely: a scoping review</article-title><source>J Clin Epidemiol</source><year>2021</year><month>11</month><volume>139</volume><fpage>287</fpage><lpage>296</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2021.05.019</pub-id><pub-id pub-id-type="medline">34091021</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blaizot</surname><given-names>A</given-names> </name><name name-style="western"><surname>Veettil</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Saidoung</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Using artificial intelligence methods for systematic review in health sciences: a systematic review</article-title><source>Res Synth Methods</source><year>2022</year><month>05</month><volume>13</volume><issue>3</issue><fpage>353</fpage><lpage>362</lpage><pub-id pub-id-type="doi">10.1002/jrsm.1553</pub-id><pub-id pub-id-type="medline">35174972</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mogoale</surname><given-names>PD</given-names> </name><name name-style="western"><surname>Pretorius</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Mogase</surname><given-names>RC</given-names> </name><name name-style="western"><surname>Segooa</surname><given-names>MA</given-names> </name></person-group><article-title>Evaluating the efficacy of AI tools in systematic literature reviews: a comprehensive analysis</article-title><source>J Inf Syst Inform</source><year>2025</year><volume>7</volume><issue>1</issue><fpage>870</fpage><lpage>888</lpage><pub-id pub-id-type="doi">10.51519/journalisi.v7i1.1035</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>van de Schoot</surname><given-names>R</given-names> </name><name name-style="western"><surname>de Bruin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schram</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Open source software for efficient and transparent reviews</article-title><comment>Preprint posted online on  Jun 22, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2006.12166</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ge</surname><given-names>L</given-names> </name><name name-style="western"><surname>Agrawal</surname><given-names>R</given-names> </name><name name-style="western"><surname>Singer</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Leveraging artificial intelligence to enhance systematic reviews in health research: advanced tools and challenges</article-title><source>Syst Rev</source><year>2024</year><month>10</month><day>25</day><volume>13</volume><issue>1</issue><fpage>269</fpage><pub-id pub-id-type="doi">10.1186/s13643-024-02682-2</pub-id><pub-id pub-id-type="medline">39456077</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>M</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Park</surname><given-names>YJ</given-names> </name><name name-style="western"><surname>Paget</surname><given-names>M</given-names> </name><name name-style="western"><surname>Naugler</surname><given-names>C</given-names> </name></person-group><article-title>Automated paper screening for clinical reviews using large language models: data analysis study</article-title><source>J Med Internet Res</source><year>2024</year><month>01</month><day>12</day><volume>26</volume><fpage>e48996</fpage><pub-id pub-id-type="doi">10.2196/48996</pub-id><pub-id pub-id-type="medline">38214966</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rathbone</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hoffmann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Glasziou</surname><given-names>P</given-names> </name></person-group><article-title>Faster title and abstract screening? Evaluating Abstrackr, a semi-automated online screening program for systematic reviewers</article-title><source>Syst Rev</source><year>2015</year><month>06</month><day>15</day><volume>4</volume><issue>1</issue><fpage>80</fpage><pub-id pub-id-type="doi">10.1186/s13643-015-0067-6</pub-id><pub-id pub-id-type="medline">26073974</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamel</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kelly</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Thavorn</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rice</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Wells</surname><given-names>GA</given-names> </name><name name-style="western"><surname>Hutton</surname><given-names>B</given-names> </name></person-group><article-title>An evaluation of DistillerSR&#x2019;s machine learning-based prioritization tool for title/abstract screening - impact on reviewer-relevant outcomes</article-title><source>BMC Med Res Methodol</source><year>2020</year><month>10</month><day>15</day><volume>20</volume><issue>1</issue><fpage>256</fpage><pub-id pub-id-type="doi">10.1186/s12874-020-01129-1</pub-id><pub-id pub-id-type="medline">33059590</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Babineau</surname><given-names>J</given-names> </name></person-group><article-title>Product review: covidence (systematic review software)</article-title><source>J Can Health Libr Assoc</source><year>2014</year><volume>35</volume><issue>2</issue><fpage>68</fpage><lpage>71</lpage><pub-id pub-id-type="doi">10.5596/c14-016</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ouzzani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hammady</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fedorowicz</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Elmagarmid</surname><given-names>A</given-names> </name></person-group><article-title>Rayyan-a web and mobile app for systematic reviews</article-title><source>Syst Rev</source><year>2016</year><month>12</month><day>5</day><volume>5</volume><issue>1</issue><fpage>210</fpage><pub-id pub-id-type="doi">10.1186/s13643-016-0384-4</pub-id><pub-id pub-id-type="medline">27919275</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Dijk</surname><given-names>SHB</given-names> </name><name name-style="western"><surname>Brusse-Keizer</surname><given-names>MGJ</given-names> </name><name name-style="western"><surname>Bucs&#x00E1;n</surname><given-names>CC</given-names> </name><name name-style="western"><surname>van der Palen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Doggen</surname><given-names>CJM</given-names> </name><name name-style="western"><surname>Lenferink</surname><given-names>A</given-names> </name></person-group><article-title>Artificial intelligence in systematic reviews: promising when appropriately used</article-title><source>BMJ Open</source><year>2023</year><month>07</month><day>7</day><volume>13</volume><issue>7</issue><fpage>e072254</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2023-072254</pub-id><pub-id pub-id-type="medline">37419641</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van der Pol</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Huizinga</surname><given-names>TW</given-names> </name><name name-style="western"><surname>Bergstra</surname><given-names>SA</given-names> </name></person-group><article-title>Is AI-assisted active learning software able to reliably speed-up systematic literature reviews in rheumatology? A real-time comparison of AI-assisted and manual abstract selection</article-title><source>RMD Open</source><year>2024</year><month>12</month><day>4</day><volume>10</volume><issue>4</issue><fpage>e005024</fpage><pub-id pub-id-type="doi">10.1136/rmdopen-2024-005024</pub-id><pub-id pub-id-type="medline">39632096</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Delgado-Chaves</surname><given-names>FM</given-names> </name><name name-style="western"><surname>Jennings</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Atalaia</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Transforming literature screening: the emerging role of large language models in systematic reviews</article-title><source>Proc Natl Acad Sci U S A</source><year>2025</year><month>01</month><day>14</day><volume>122</volume><issue>2</issue><fpage>e2411962122</fpage><pub-id pub-id-type="doi">10.1073/pnas.2411962122</pub-id><pub-id pub-id-type="medline">39761403</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Akaraci</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Tate</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Evaluating the use of artificial intelligence (AI) in systematic review abstract screening: a comparative study of AI-aided tools</article-title><source>Research Square</source><comment>Preprint posted online on  Jan 21, 2026</comment><pub-id pub-id-type="doi">10.21203/rs.3.rs-7915337/v1</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gauthier Mongeon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ouadfel</surname><given-names>S</given-names> </name><name name-style="western"><surname>Thullier</surname><given-names>F</given-names> </name><name name-style="western"><surname>Gaboury</surname><given-names>S</given-names> </name><name name-style="western"><surname>Arsenault-Lapierre</surname><given-names>G</given-names> </name></person-group><article-title>Manual versus AI-assisted document screening (ASReview): a comparative analysis within a rapid systematized review in the social sciences</article-title><source>Int J Soc Res Methodol</source><year>2026</year><fpage>1</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.1080/13645579.2026.2668099</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kataoka</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Banno</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kyo</surname><given-names>M</given-names> </name><etal/></person-group><article-title>TiAb review plugin: a browser-based tool for AI-assisted title and abstract screening</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 8, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2604.08602</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Boesen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hemkens</surname><given-names>L</given-names> </name><name name-style="western"><surname>Janiaud</surname><given-names>P</given-names> </name><name name-style="western"><surname>Hirt</surname><given-names>J</given-names> </name></person-group><article-title>Machine-learning assisted screening for evidence synthesis: a case study of using the asreview tool</article-title><source>SSRN</source><comment>Preprint posted online on  Feb 6, 2025</comment><pub-id pub-id-type="doi">10.2139/ssrn.5125681</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Oude Wolcherink</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Pouwels</surname><given-names>XGLV</given-names> </name><name name-style="western"><surname>van Dijk</surname><given-names>SHB</given-names> </name><name name-style="western"><surname>Doggen</surname><given-names>CJM</given-names> </name><name name-style="western"><surname>Koffijberg</surname><given-names>H</given-names> </name></person-group><article-title>Can artificial intelligence separate the wheat from the chaff in systematic reviews of health economic articles?</article-title><source>Expert Rev Pharmacoecon Outcomes Res</source><year>2023</year><volume>23</volume><issue>9</issue><fpage>1049</fpage><lpage>1056</lpage><pub-id pub-id-type="doi">10.1080/14737167.2023.2234639</pub-id><pub-id pub-id-type="medline">37573521</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chan</surname><given-names>YT</given-names> </name><name name-style="western"><surname>Abad</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Dibart</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kernitsky</surname><given-names>JR</given-names> </name></person-group><article-title>Assessing the article screening efficiency of artificial intelligence for systematic reviews</article-title><source>J Dent</source><year>2024</year><month>10</month><volume>149</volume><fpage>105259</fpage><pub-id pub-id-type="doi">10.1016/j.jdent.2024.105259</pub-id><pub-id pub-id-type="medline">39067652</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Scherhag</surname><given-names>J</given-names> </name><name name-style="western"><surname>Burgard</surname><given-names>T</given-names> </name></person-group><article-title>Performance of semi-automated screening using rayyan and asreview: a retrospective analysis of potential work reduction and different stopping rules</article-title><source>PsychArchives</source><comment>Preprint posted online on  May 3, 2023</comment><pub-id pub-id-type="doi">10.23668/psycharchives.12843</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Teijema</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Ribeiro</surname><given-names>G</given-names> </name><name name-style="western"><surname>Seuren</surname><given-names>S</given-names> </name><name name-style="western"><surname>Anadria</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bagheri</surname><given-names>A</given-names> </name><name name-style="western"><surname>van de Schoot</surname><given-names>R</given-names> </name></person-group><article-title>Simulation-based active learning for systematic reviews: a scoping review of literature</article-title><source>J Inf Sci</source><year>2025</year><fpage>01655515251379058</fpage><pub-id pub-id-type="doi">10.1177/01655515251379058</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><article-title>Navigating the maze of models in ASReview</article-title><source>ASReview</source><year>2025</year><access-date>2026-07-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://asreview.nl/blog/asreview-model-selection-guide/">https://asreview.nl/blog/asreview-model-selection-guide/</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boetje</surname><given-names>J</given-names> </name><name name-style="western"><surname>van de Schoot</surname><given-names>R</given-names> </name></person-group><article-title>The SAFE procedure: a practical stopping heuristic for active learning-based screening in systematic reviews and meta-analyses</article-title><source>Syst Rev</source><year>2024</year><month>03</month><day>1</day><volume>13</volume><issue>1</issue><fpage>81</fpage><pub-id pub-id-type="doi">10.1186/s13643-024-02502-7</pub-id><pub-id pub-id-type="medline">38429798</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steijger</surname><given-names>D</given-names> </name><name name-style="western"><surname>Christie</surname><given-names>H</given-names> </name><name name-style="western"><surname>Aarts</surname><given-names>S</given-names> </name><name name-style="western"><surname>IJselsteijn</surname><given-names>W</given-names> </name><name name-style="western"><surname>Verbeek</surname><given-names>H</given-names> </name><name name-style="western"><surname>de Vugt</surname><given-names>M</given-names> </name></person-group><article-title>Use of artificial intelligence to support quality of life of people with dementia: a scoping review</article-title><source>Ageing Res Rev</source><year>2025</year><month>06</month><volume>108</volume><fpage>102741</fpage><pub-id pub-id-type="doi">10.1016/j.arr.2025.102741</pub-id><pub-id pub-id-type="medline">40188991</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Holst</surname><given-names>D</given-names> </name><name name-style="western"><surname>Moenck</surname><given-names>K</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schmedemann</surname><given-names>O</given-names> </name><name name-style="western"><surname>Sch&#x00FC;ppstuhl</surname><given-names>T</given-names> </name></person-group><article-title>Transparent reporting of AI in systematic literature reviews: development of the PRISMA-trAIce checklist</article-title><source>JMIR AI</source><year>2025</year><month>12</month><day>10</day><volume>4</volume><fpage>e80247</fpage><pub-id pub-id-type="doi">10.2196/80247</pub-id><pub-id pub-id-type="medline">41370833</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Salton</surname><given-names>G</given-names> </name><name name-style="western"><surname>Buckley</surname><given-names>C</given-names> </name></person-group><article-title>Term-weighting approaches in automatic text retrieval</article-title><source>Inf Process Manag</source><year>1988</year><month>01</month><volume>24</volume><issue>5</issue><fpage>513</fpage><lpage>523</lpage><pub-id pub-id-type="doi">10.1016/0306-4573(88)90021-0</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>McCallum</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nigam</surname><given-names>K</given-names> </name></person-group><article-title>A comparison of event models for naive bayes text classification</article-title><access-date>2026-07-22</access-date><conf-name>AAAI-98 Workshop on Learning for Text Categorization</conf-name><conf-date>Jul 27, 1998</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://cdn.aaai.org/Workshops/1998/WS-98-05/WS98-05-007.pdf">https://cdn.aaai.org/Workshops/1998/WS-98-05/WS98-05-007.pdf</ext-link></comment></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van de Schoot</surname><given-names>R</given-names> </name><name name-style="western"><surname>de Bruin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schram</surname><given-names>R</given-names> </name><etal/></person-group><article-title>An open source machine learning framework for efficient and transparent systematic reviews</article-title><source>Nat Mach Intell</source><year>2021</year><volume>3</volume><issue>2</issue><fpage>125</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1038/s42256-020-00287-7</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><article-title>A coefficient of agreement for nominal scales</article-title><source>Educ Psychol Meas</source><year>1960</year><month>04</month><volume>20</volume><issue>1</issue><fpage>37</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1177/001316446002000104</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McNEMAR</surname><given-names>Q</given-names> </name></person-group><article-title>Note on the sampling error of the difference between correlated proportions or percentages</article-title><source>Psychometrika</source><year>1947</year><month>06</month><volume>12</volume><issue>2</issue><fpage>153</fpage><lpage>157</lpage><pub-id pub-id-type="doi">10.1007/BF02295996</pub-id><pub-id pub-id-type="medline">20254758</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Campos</surname><given-names>DG</given-names> </name><name name-style="western"><surname>F&#x00FC;tterer</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gfr&#x00F6;rer</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Screening smarter, not harder: a comparative analysis of machine learning screening algorithms and heuristic stopping criteria for systematic reviews in educational research</article-title><source>Educ Psychol Rev</source><year>2024</year><month>03</month><volume>36</volume><issue>1</issue><fpage>19</fpage><pub-id pub-id-type="doi">10.1007/s10648-024-09862-5</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>LD</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>TT</given-names> </name><name name-style="western"><surname>DasGupta</surname><given-names>A</given-names> </name></person-group><article-title>Interval estimation for a binomial proportion</article-title><source>Statist Sci</source><year>2001</year><volume>16</volume><issue>2</issue><fpage>101</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1214/ss/1009213286</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rai</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tabata</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sasaki</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Expanding the feasibility of systematic reviews with AI support: a practical case using ASReview</article-title><source>Igaku Toshokan</source><year>2025</year><volume>72</volume><issue>3</issue><fpage>136</fpage><lpage>141</lpage><pub-id pub-id-type="doi">10.7142/igakutoshokan.72.3_136</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>MV</given-names> </name><name name-style="western"><surname>Su</surname><given-names>E</given-names> </name><name name-style="western"><surname>Flores Miranda</surname><given-names>A</given-names> </name><name name-style="western"><surname>Saha</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sussman</surname><given-names>J</given-names> </name></person-group><article-title>Evaluating the efficacy of artificial intelligence tools for the automation of systematic reviews in cancer research: a systematic review</article-title><source>Cancer Epidemiol</source><year>2024</year><month>02</month><volume>88</volume><fpage>102511</fpage><pub-id pub-id-type="doi">10.1016/j.canep.2023.102511</pub-id><pub-id pub-id-type="medline">38071872</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Teijema</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>de Bruin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bagheri</surname><given-names>A</given-names> </name><name name-style="western"><surname>van de Schoot</surname><given-names>R</given-names> </name></person-group><article-title>Large-scale simulation study of active learning models for systematic reviews</article-title><source>Int J Data Sci Anal</source><year>2025</year><month>11</month><volume>20</volume><issue>6</issue><fpage>5435</fpage><lpage>5456</lpage><pub-id pub-id-type="doi">10.1007/s41060-025-00777-0</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kempny</surname><given-names>C</given-names> </name><name name-style="western"><surname>Annac</surname><given-names>K</given-names> </name><name name-style="western"><surname>Wahidie</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yilmaz-Aslan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Brzoska</surname><given-names>P</given-names> </name></person-group><article-title>When to stop reviewing: validation of stop criteria in ASReview</article-title><source>BMC Med Res Methodol</source><year>2026</year><month>05</month><day>9</day><volume>26</volume><issue>1</issue><fpage>109</fpage><pub-id pub-id-type="doi">10.1186/s12874-026-02866-5</pub-id><pub-id pub-id-type="medline">42106619</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>O&#x2019;Mara-Eves</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>J</given-names> </name><name name-style="western"><surname>McNaught</surname><given-names>J</given-names> </name><name name-style="western"><surname>Miwa</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name></person-group><article-title>Using text mining for study identification in systematic reviews: a systematic review of current approaches</article-title><source>Syst Rev</source><year>2015</year><month>01</month><day>14</day><volume>4</volume><issue>1</issue><fpage>5</fpage><pub-id pub-id-type="doi">10.1186/2046-4053-4-5</pub-id><pub-id pub-id-type="medline">25588314</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sanghera</surname><given-names>R</given-names> </name><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>El Khoury</surname><given-names>M</given-names> </name><etal/></person-group><article-title>High-performance automated abstract screening with large language model ensembles</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>05</month><day>1</day><volume>32</volume><issue>5</issue><fpage>893</fpage><lpage>904</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf050</pub-id><pub-id pub-id-type="medline">40119675</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cumpston</surname><given-names>M</given-names> </name><name name-style="western"><surname>Li</surname><given-names>T</given-names> </name><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>Updated guidance for trusted systematic reviews: a new edition of the Cochrane Handbook for Systematic Reviews of Interventions</article-title><source>Cochrane Database Syst Rev</source><year>2019</year><month>10</month><day>3</day><volume>10</volume><issue>10</issue><fpage>ED000142</fpage><pub-id pub-id-type="doi">10.1002/14651858.ED000142</pub-id><pub-id pub-id-type="medline">31643080</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gartlehner</surname><given-names>G</given-names> </name><name name-style="western"><surname>Affengruber</surname><given-names>L</given-names> </name><name name-style="western"><surname>Titscher</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Single-reviewer abstract screening missed 13 percent of relevant studies: a crowd-based, randomized controlled trial</article-title><source>J Clin Epidemiol</source><year>2020</year><month>05</month><volume>121</volume><fpage>20</fpage><lpage>28</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2020.01.005</pub-id><pub-id pub-id-type="medline">31972274</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>An application of hierarchical kappa-type statistics in the assessment of majority agreement among multiple observers</article-title><source>Biometrics</source><year>1977</year><month>06</month><volume>33</volume><issue>2</issue><fpage>363</fpage><lpage>374</lpage><pub-id pub-id-type="medline">884196</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zec</surname><given-names>S</given-names> </name><name name-style="western"><surname>Soriani</surname><given-names>N</given-names> </name><name name-style="western"><surname>Comoretto</surname><given-names>R</given-names> </name><name name-style="western"><surname>Baldi</surname><given-names>I</given-names> </name></person-group><article-title>High agreement and high prevalence: the paradox of Cohen&#x2019;s kappa</article-title><source>Open Nurs J</source><year>2017</year><volume>11</volume><fpage>211</fpage><lpage>218</lpage><pub-id pub-id-type="doi">10.2174/1874434601711010211</pub-id><pub-id pub-id-type="medline">29238424</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Agresti</surname><given-names>A</given-names> </name></person-group><source>Categorical Data Analysis</source><year>2013</year><publisher-name>John Wiley &#x0026; Sons</publisher-name><pub-id pub-id-type="other">978-0-470-46363-5</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Flemyng</surname><given-names>E</given-names> </name><name name-style="western"><surname>Noel-Storr</surname><given-names>A</given-names> </name><name name-style="western"><surname>Macura</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Position statement on artificial intelligence (AI) use in evidence synthesis across Cochrane, the Campbell Collaboration, JBI and the Collaboration for Environmental Evidence 2025</article-title><source>Environ Evid</source><year>2025</year><volume>14</volume><issue>1</issue><pub-id pub-id-type="doi">10.1186/s13750-025-00374-5</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferdinands</surname><given-names>G</given-names> </name><name name-style="western"><surname>Schram</surname><given-names>R</given-names> </name><name name-style="western"><surname>de Bruin</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Performance of active learning models for screening prioritization in systematic reviews: a simulation study into the average time to discover relevant records</article-title><source>Syst Rev</source><year>2023</year><month>06</month><day>20</day><volume>12</volume><issue>1</issue><fpage>100</fpage><pub-id pub-id-type="doi">10.1186/s13643-023-02257-7</pub-id><pub-id pub-id-type="medline">37340494</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Source datasets and screening decision files for the comparative evaluation of 2 AI-assisted screening approaches within ASReview compared with manual screening.</p><media xlink:href="formative_v10i1e84989_app1.zip" xlink:title="ZIP File, 12375 KB"/></supplementary-material></app-group></back></article>