<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e88406</article-id><article-id pub-id-type="doi">10.2196/88406</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>A Systematic Evaluation of Cohort Selection Criteria and Their Impact on Machine Learning Model Performance and Demographic Disparities in COVID-19 Outcomes: Cohort Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Haghighathoseini</surname><given-names>Atefehsadat</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wojtusiak</surname><given-names>Janusz</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Min</surname><given-names>Hua</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>M Menon</surname><given-names>Nirup</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Health Administration, Policy and Informatics, George Mason University</institution><addr-line>4400 University Dr</addr-line><addr-line>Fairfax</addr-line><addr-line>VA</addr-line><country>United States</country></aff><aff id="aff2"><institution>Information Systems and Operations Management, Costello College of Business, George Mason University</institution><addr-line>Fairfax</addr-line><addr-line>VA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Taiwo</surname><given-names>Peter</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Atefehsadat Haghighathoseini, PhD, Department of Health Administration, Policy and Informatics, George Mason University, 4400 University Dr, Fairfax, VA, 22030, United States, +1 (703) 993-1929; <email>ahoseini@gmu.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>20</day><month>8</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e88406</elocation-id><history><date date-type="received"><day>24</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>10</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>11</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Atefehsadat Haghighathoseini, Janusz Wojtusiak, Hua Min, Nirup M Menon. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 20.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e88406"/><abstract><sec><title>Background</title><p>Cohort selection criteria play a critical role in shaping machine learning (ML) model performance and the equity of clinical outcome predictions across demographic groups. In practice, cohort definitions are often influenced by variable and sometimes inconsistent data processing decisions, which may introduce bias and limit the generalizability of ML models. During the COVID-19 pandemic, rapid cohort construction further increased concerns about transparency and fairness in ML-based analyses.</p></sec><sec><title>Objective</title><p>This study aimed to systematically examine how cohort selection and data processing decisions influence ML performance and demographic equity in predicting COVID-19&#x2013;related in-hospital mortality.</p></sec><sec sec-type="methods"><title>Methods</title><p>Using data from the National COVID Cohort Collaborative (N3C), we evaluated 2 sets of cohorts. Set 1 consisted of 16 cohorts derived from 4 primary data processing decisions, including COVID-19 case identification, inpatient inclusion, diagnosis date selection, and admission timestamp availability. Set 2 expanded this design to 64 cohorts by additionally applying provider ID and location identifier filtering. Model performance was assessed using the area under the receiver operating characteristic curve (AUC) across multiple training-testing cohort combinations. Three ML models&#x2014;logistic regression, random forest, and gradient boosting&#x2014;were evaluated using 3 analytical approaches: maximum AUC classification, direct AUC regression, and AUC gap analysis. Performance was further examined across demographic subgroups defined by gender, race, and ethnicity.</p></sec><sec sec-type="results"><title>Results</title><p>This study analyzed data from the N3C, including patients with a first positive COVID-19 diagnosis between August 1, 2020, and December 31, 2021. Data preprocessing, cohort construction, and model development were completed prior to analysis. Cohort selection decisions had a substantial impact on ML model performance. Admission time inclusion or exclusion emerged as the most influential factor in Set 1 and consistently affected model accuracy across analytical approaches. In Set 2, this decision remained important, while additional criteria, particularly provider ID filtering, also significantly influenced results. The importance of specific decisions varied across models and evaluation strategies. Analyses across demographic subgroups showed that data processing decisions affected predictive performance differently by gender, race, and ethnicity.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Seemingly minor cohort selection and data processing decisions can meaningfully affect both predictive accuracy and demographic equity in ML-based COVID-19 outcome prediction. These findings highlight the risk of bias introduced by differences in cohort definitions and underscore the need for transparent, standardized, and equity-aware cohort selection practices to support fair and reproducible ML research in health care.</p></sec></abstract><kwd-group><kwd>Cohort selection</kwd><kwd>machine learning</kwd><kwd>COVID-19</kwd><kwd>model performance</kwd><kwd>demographic disparities</kwd><kwd>data processing decisions</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The integration of machine learning (ML) in health care, particularly for clinical decision-making in diagnosis, prognosis, and risk prediction, is advancing rapidly [<xref ref-type="bibr" rid="ref1">1</xref>]. ML&#x2019;s ability to efficiently analyze large, complex datasets has enhanced traditional clinical reasoning and enabled precision medicine, where treatments are tailored to individual patient characteristics [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. However, the performance and generalizability of ML models depend on the quality and structure of the data [<xref ref-type="bibr" rid="ref4">4</xref>], underscoring the importance of thoughtful cohort selection [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Most ML studies evaluate generalizability using a test set; yet, the test set is selected after data processing decisions have been made, thus potentially limiting real generalizability. Cohort selection&#x2014;defining which patients are included in a study based on diagnosis codes, treatment timelines, and other criteria&#x2014;is a foundational step in ML model development. Poorly defined or nontransparent criteria can introduce bias [<xref ref-type="bibr" rid="ref7">7</xref>], produce heterogeneous cohorts, and compromise model validity [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Real-world data sources like electronic health records (EHRs) offer rich clinical information but also present challenges, including missing data, inconsistent coding, and variation across institutions, all of which affect model development and reproducibility [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>Inconsistencies in cohort construction practices and the absence of standardized guidelines make it difficult to compare studies or validate ML models across settings [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. The use of proxy definitions without adequate clinical validation can distort clinical characteristics, while underrepresentation of racial and ethnic minority groups [<xref ref-type="bibr" rid="ref12">12</xref>] can skew outcomes and reinforce health disparities [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Such limitations can impair a model&#x2019;s ability to generalize to diverse patient populations, reducing its clinical utility [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>Selection bias in ML carries significant implications. Inadequately defined cohorts can lead to overfitting, limited real-world applicability, and unequal model performance across subpopulations. For example, disparities in the allocation of advanced heart failure therapies by race and gender reveal how implicit bias in data and decision-making can affect outcomes [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. Addressing these challenges requires a structured and transparent approach to cohort definition that accounts for equity, clinical relevance, and methodological rigor [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>Several strategies have been proposed to improve ML practice in this context. These include advanced data preprocessing, interpretability frameworks, and performance metrics that reflect the specific characteristics of the cohort under study [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Collaboration between clinicians and data scientists is essential to ensure that cohort definitions are clinically appropriate and that ML models align with patient-centered goals and ethical standards [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>].</p><p>The COVID-19 pandemic provided a critical test case for ML deployment in health care, particularly through large-scale data initiatives like the National COVID Cohort Collaborative (N3C). These resources enabled broad analysis of patient outcomes and risk factors [<xref ref-type="bibr" rid="ref24">24</xref>]. However, the urgency of pandemic response often led to expedited cohort construction with less rigorous inclusion criteria [<xref ref-type="bibr" rid="ref25">25</xref>], raising concerns about the generalizability and fairness of resulting models [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. Vulnerable populations were disproportionately affected, and inconsistent or arbitrary cohort definitions sometimes reduced predictive accuracy for marginalized groups [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. This context highlights the urgent need for standardized, explicit cohort selection criteria to ensure ML models are valid, equitable, and reproducible [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Moving forward, the development of comprehensive guidelines for cohort construction, along with validation of existing models against diverse benchmarks, will be critical for achieving trustworthy and impactful ML applications in health care [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>].</p><p>To address these gaps&#x2014;including the lack of standardized cohort construction criteria [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>] and the limited validation of models against diverse benchmarks [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]&#x2014;this study presents a comprehensive examination of how varying cohort construction decisions influence the performance and fairness of ML models in predicting COVID-19 outcomes. By systematically analyzing combinations of data processing criteria across multiple modeling approaches and evaluating their impact on both overall accuracy and subgroup consistency, this study offers insights into the unintended consequences of cohort definition choices. The findings aim to inform more transparent and equitable practices in ML-based health care research, emphasizing the importance of deliberate, well-documented cohort selection processes to support model reproducibility, reliability, and inclusivity.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Data Source</title><p>The N3C offers the most extensive harmonized repository of COVID-19&#x2013;related patient data in the United States, integrating EHRs from more than 70 health care institutions nationwide [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. This centralized initiative provides a unified data platform through the N3C Data Enclave, enabling researchers to conduct large-scale, high-quality analyses focused on COVID-19 outcomes, treatment effectiveness, and health care disparities [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. This study used the limited dataset (LDS) available through the N3C, which contains individual-level patient data with direct identifiers removed. This approach maintains patients&#x2019; privacy while preserving the detailed clinical information necessary for research [<xref ref-type="bibr" rid="ref38">38</xref>]. The dataset conforms to the Observational Medical Outcomes Partnership (OMOP) Common Data Model (CDM), which standardizes clinical data structures across contributing institutions to support consistency and reproducibility [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. For inpatient data analysis, the N3C employs the concept of a &#x201C;Macro-visit,&#x201D; encompassing hospital admissions, observation periods, extended multiday stays following outpatient procedures, emergency department visits, and overlapping outpatient or telehealth encounters that occurred during the COVID-19 pandemic [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref40">40</xref>].</p><p>The primary objective of this research is to predict outcomes for patients hospitalized with COVID-19, with a specific focus on in-hospital mortality. The unit of analysis is the patient hospitalization. To support this objective, the data must be transformed so that each hospital stay is represented by a single, consolidated record&#x2014;commonly referred to as a flat table or analytic file&#x2014;which is a typical structure for applying ML methods.</p></sec><sec id="s2-2"><title>Data Preprocessing Decisions</title><sec id="s2-2-1"><title>Overview</title><p>One of the most critical steps in any data analysis, including building ML models, is cohort construction. In this study, a series of reasonable but variable data processing decisions were applied to approximately 22 million patients from the N3C dataset.</p><p>The study distinguishes between 4 sets of cohort-defining decisions. The first set comprises 4 primary data processing decisions (decisions 1-4) that result in 16 distinct cohorts (referred to as Set 1). These decisions include the identification of COVID-19 cases, identification of inpatient hospitalization records, identification of COVID-19&#x2013;related hospitalizations, and potential exclusion of records with missing data on the specific time of admission. These different but plausible design choices can lead to the creation of up to 16 different datasets, each varying in size and characteristics. By incorporating 4 additional decisions&#x2014;filtering by provider IDs and location identifiers (decisions 5 and 6)&#x2014;into the primary set of decisions, the number of possible cohorts expands to 64 (referred to as Set 2). The details of these decisions are provided below for both Set 1 and Set 2.</p><p>The analysis of these cohorts revealed statistically significant differences in both demographic and outcome variables, underscoring the impact of these preprocessing decisions on cohort composition. Specifically, significant variations were observed in age, gender, race, ethnicity, and state, as well as in length of stay and survival (expiry flag). Consistently small <italic>P</italic> values (&#x003C;.001) indicate that these differences are unlikely to be due to chance, indicating that cohort membership is strongly associated with survival outcomes and demographic characteristics. To evaluate these differences, multiple statistical tests were applied, including the chi-square test for categorical variables (gender, race, ethnicity, state, and expiry flag), with df reported for each; the independent 2-sample <italic>t</italic> test (Welch <italic>t</italic> test) for continuous variables (age and length of stay), as a parametric test assuming unequal variances; and the Mann-Whitney <italic>U</italic> test as a nonparametric alternative for the same continuous variables. Together, these results demonstrate that the seemingly minor preprocessing choices substantially influence the resulting cohorts, ultimately shaping both the composition and the outcomes observed in the study.</p></sec><sec id="s2-2-2"><title>Set 1: Main Decisions</title><sec id="s2-2-2-1"><title>Decision A</title><p>Identifying patients with COVID-19: there have been a variety of tests and options for identifying patients with COVID-19. This study considers that some patients have a confirmed positive COVID-19 laboratory test result, while others are identified based on diagnostic codes [<xref ref-type="bibr" rid="ref36">36</xref>].</p></sec><sec id="s2-2-2-2"><title>Decision B</title><p>Identifying hospitalization records among patient encounters: 2 approaches were considered for identifying hospitalization records. One approach involved performing a wildcard text search for visit concepts containing terms such as &#x201C;inpatient,&#x201D; &#x201C;observ,&#x201D; and &#x201C;hospital.&#x201D; The other relied on an N3C-defined variable for hospitalization status, which is calculated by the N3C team based on clinical data.</p></sec><sec id="s2-2-2-3"><title>Decision C</title><p>A time window is considered for a COVID-19 positive test to be related to the macro-visit. This study compared a 7-day versus a 10-day window prior to the start of the inpatient record, extending through the second day of hospitalization.</p></sec><sec id="s2-2-2-4"><title>Decision D</title><p>The final decision relates to how admission timing is handled during data processing. One approach was to include only records with an exact admission time, thus allowing an exact 48-hour observation window used in subsequent analyses. However, excluding records without an exact time stamp will result in less generalizable results. The other extreme is including all records, which would require estimating the 2-day hospitalization period based on the admission and discharge date alone. Moreover, this approach does not capture the 48-hour window, as it is often calculated from midnight to midnight.</p></sec></sec></sec><sec id="s2-3"><title>Set 2: Additional Decisions</title><sec id="s2-3-1"><title>Decision E</title><p>Keep only records with provider ID (all encounters have a known provider) or include all records (with or without provider ID) to preserve the entire cohort.</p></sec><sec id="s2-3-2"><title>Decision F</title><p>One approach is to drop the records without a location identifier. The other approach is to keep all the records, whether they have a location identifier, which preserves cohort size but introduces uncertainty to any analysis at the location level.</p></sec></sec><sec id="s2-4"><title>Other Decisions</title><p>Data preprocessing contains other decisions that were not investigated for the simplicity of the presented work.</p><p>After decision C, data were filtered to include only patients whose first positive COVID-19 diagnosis occurred between August 1, 2020, and December 31, 2021. This period was selected to avoid the earliest phase of the COVID-19 pandemic, when testing availability, diagnostic coding practices, and clinical management protocols were rapidly evolving [<xref ref-type="bibr" rid="ref25">25</xref>]. Restricting the analysis to this timeframe provides a more consistent basis for cohort comparison while maintaining a large and diverse patient population. Similar periods have been used in prior N3C studies of COVID-19 outcomes [<xref ref-type="bibr" rid="ref36">36</xref>]. Then, hospitalization data were merged with other core data to be used in the final analysis with a unique identifier and variables including gender, date of birth, race, ethnicity, and age at death. Patients younger than 18 years were excluded from all cohorts. The analysis was restricted to adults aged 18 years and older, consistent with the standard definition of adulthood used in clinical research and prior N3C studies of COVID-19 outcomes [<xref ref-type="bibr" rid="ref36">36</xref>]. Pediatric patients differ from adults in disease presentation, hospitalization patterns, and clinical outcomes. Restricting the study population to adults therefore provides a more clinically comparable population for evaluating the impact of cohort construction decisions [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>].</p></sec><sec id="s2-5"><title>Study Design</title><sec id="s2-5-1"><title>Study Design Overview</title><p>The study is structured to evaluate the impact of cohort selection criteria on ML model performance and demographic equity in predicting COVID-19 outcomes. The experiment is divided into 3 main approaches, each focusing on different aspects of model evaluation and data processing decisions. As shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>, the flowchart illustrates the experimental design of this study for analyzing the impact of data processing decisions on ML model performance in COVID-19 outcomes.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flowchart illustrating experimental design for analyzing the impact of data processing decisions on machine learning (ML) model performance in COVID-19 outcomes. AUC: area under the receiver operating characteristic curve; ML: machine learning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88406_fig01.png"/></fig><p>Race, ethnicity, and gender were harmonized using rule-based recoding of the original N3C concept labels into analytically consistent categories. Sex was categorized as female, male, or unknown, with unmatched or ambiguous values (eg, no matching concept, sex unknown, other) grouped as unknown. Race categories were aggregated into Black, White, Asian, Multi_Race, Native Hawaiian, and unknown. Specifically, Black or African American and Black were grouped as Black, while multiple Asian subcategories (eg, Asian, Asian Indian, Filipino, Chinese, Korean, Japanese, and Vietnamese) were consolidated into Asian. Categories such as multiple race, multiple races, and more than one race were grouped as Multi_Race, and Native Hawaiian or Other Pacific Islander was retained as Native Hawaiian. Race values that were missing, unmapped, or ambiguous (eg, no matching concept, unknown, and no information) were assigned to unknown. Ethnicity was harmonized into Hispanic, non-Hispanic, and unknown, with Hispanic or Latino mapped to Hispanic, Not Hispanic or Latino mapped to non-Hispanic, and missing or ambiguous values grouped as unknown. These harmonized categories were used for all descriptive summaries and subgroup analyses.</p><p>The study involves the systematic extraction of relevant clinical measurements and medical device usage data for patients within defined cohorts from the N3C dataset. Measurements are organized based on their unique concept identifiers, selected from the concept set member&#x2019;s table. These measurements cover a wide range of clinically important indicators, including, but not limited to, respiratory rate, glomerular filtration rate (GFR), cardiac troponin T, sodium, ABG indices, prothrombin time, alanine transaminase, calcium, systolic blood pressure (BP), heart rate, CD3+ CD8+ T cells, albumin, erythrocyte sedimentation rate, venous lactate, BMI, complete blood count (CBC) with platelets, CD3+ CD4+ T cells, cardiac troponin I, interleukin 6, fraction of inspired oxygen, blood urea nitrogen, temperature, bilirubin, D-dimer, alkaline phosphatase, fibrinogen, creatinine, chloride, glucose, C-reactive protein, urine output, N-terminal pro&#x2013;B-type natriuretic peptide (NT-proBNP), diastolic BP, weight, ferritin, and interleukin 10.</p><p>In addition, the study incorporates data on a broad range of medical devices used during hospitalization, represented as binary indicators reflecting their presence or absence. These devices include endotracheal tube, peritoneal hemodialysis, imaging-related, medical supplies, oxygen delivery systems, respiratory therapy equipment, room air, skin care, surgical related, transfusion related, and ventilator, and categories labeled other or unknown. Device categories were defined through clinical review and inventory-based grouping of concept names according to their functional and clinical relevance.</p><p>This combined extraction and categorization focuses on the first 48 hours of hospital admission, capturing the acute phase of patient care. The resulting structured dataset of clinical measurements and device use provides a comprehensive and organized representation of patient status during this critical early period of hospitalization. This structured dataset served as the basis for subsequent ML analyses, including data partitioning, missing data handling, and model development as described below.</p><p>For each cohort, the dataset was partitioned into training and testing subsets using a deterministic hash-based approach applied to the patient identifier, resulting in approximately 80% of observations assigned to the training set and 20% assigned to the testing set. The outcome variable was defined as in-hospital mortality, and all remaining variables after cohort construction and preprocessing were used as input features. Missing data were addressed using mode imputation. Specifically, for each variable, the most frequently observed value in the training set was used to replace missing observations, and the same training-derived imputation value was subsequently applied to the corresponding testing set. This approach ensured that no information from the testing data was used during preprocessing or imputation. No additional feature-selection procedure was performed, as the primary objective was to evaluate the impact of cohort construction decisions rather than optimize model performance. Logistic regression (LR) and random forest (RF) models incorporated class-weight balancing to address class imbalance. Model hyperparameters were predefined and held constant across all cohorts to ensure comparability. LR was implemented with a maximum of 1000 iterations and balanced class weights; RF used 1000 trees with a maximum depth of 12 and balanced class weights; and gradient boosting (GB) used 150 estimators, a learning rate of 0.5, and a maximum depth of 5. No information from the testing datasets was used during feature construction, model training, or model selection.</p><p>Three ML models were implemented, including RF, GB, and L2-regularized LR [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. Standard implementations of these models were used, with class balancing applied where appropriate. Hyperparameters were specified a priori and were not further tuned, as the primary objective of this study was to evaluate the impact of cohort construction decisions rather than to optimize model performance.</p><p>Cross-validation was not applied in this study. While cross-validation is commonly used to estimate model performance within a single dataset, the primary objective of this work is to evaluate the impact of cohort construction decisions across different cohort definitions. Therefore, a cross-cohort training-testing framework was used, in which models trained on one cohort were evaluated on multiple alternative cohorts. This design enables assessment of generalizability and the influence of data processing decisions, rather than model optimization. All analyses were conducted using standard ML libraries, and the deterministic data partitioning approach ensured reproducibility of results across experiments.</p><p>To provide a more comprehensive evaluation beyond discrimination performance, fairness, and calibration analyses were conducted. Fairness was assessed using demographic parity and equality of opportunity, which quantify disparities in model predictions across demographic subgroups. These metrics were computed for gender, race, ethnicity, and age groups across all cohorts and models. In addition to fairness metrics, model calibration was evaluated to assess the agreement between predicted probabilities and observed outcomes. Calibration curves were generated for each model across all training-testing cohort combinations. This analysis was performed separately for Set 1 (16 cohorts) and Set 2 (64 cohorts), allowing for assessment of prediction reliability and generalizability under different cohort construction scenarios.</p><p>To systematically investigate these effects, the study is organized into three complementary experimental components: (1) model performance evaluation, which examines predictive accuracy across cohorts; (2) cross-cohort analysis of data processing decisions, which identifies the influence of cohort construction choices on model performance; and (3) stratified analysis across demographic groups, which evaluates how these decisions affect performance across subpopulations. Each component is described in detail below.</p><p>To ensure full transparency and reproducibility, detailed numerical results are provided in the supplementary materials. <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> presents cohort-level summaries, including sample sizes, training, and testing splits, and demographic distributions at both the row and patient levels across the 16- and 64-cohort configurations. <xref ref-type="supplementary-material" rid="app2">Multimedia Appendices 2</xref> and <xref ref-type="supplementary-material" rid="app3">3</xref> report complete model performance metrics (area under the receiver operating characteristic curve [AUC], precision, recall, and <italic>F</italic><sub>1</sub>-score) for all models across the 16 and 64 cohorts, respectively. <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> provides subgroup-specific AUC results stratified by gender, race, and ethnicity across all cohorts. Due to space constraints, the main manuscript presents representative results and key trends, while the appendices provide the full set of numerical results for comprehensive evaluation.</p></sec><sec id="s2-5-2"><title>Experiment 1: Model Accuracy</title><p>Standard supervised ML models were trained on the training set, and the predicted outcome was survival at the end of hospitalization. RFs [<xref ref-type="bibr" rid="ref43">43</xref>], GB [<xref ref-type="bibr" rid="ref43">43</xref>], and L2-regularized LR [<xref ref-type="bibr" rid="ref44">44</xref>] models were trained [<xref ref-type="bibr" rid="ref41">41</xref>]. Metrics are important for evaluating the performance of ML models [<xref ref-type="bibr" rid="ref45">45</xref>]. They provide insights into how well a model&#x2019;s predictions match the data. This study uses several metrics, including AUC, precision, recall, and <italic>F</italic><sub>1</sub>-score. For each cohort in Set 1, metrics are computed across all training-testing combinations (eg, training on Cohort 1 and testing on Cohorts 1&#x2010;16, training on Cohort 2 and testing on Cohorts 1&#x2010;16, etc). The same procedure is applied to all cohort pairs in Set 2. It is important to evaluate ML models across subpopulations defined by sex, race, and ethnicity to test fairness and reliability. Subpopulations based on sex include males and females, while subpopulations based on race include White, Black, Asian, Native Hawaiian, and multiple races. Subpopulations based on ethnicity include Hispanic and non-Hispanic.</p><p>To assess model discrimination overall and across demographic subgroups, the performance of the ML models in predicting survival at the end of hospitalization was evaluated using the AUC.</p><p>This evaluation also allows for auditing fairness both in the overall population and within subgroups. AUC is a common evaluation metric for binary classification models that measures the probability of a randomly selected positive instance being ranked higher than a random negative instance across all possible thresholds. The best-performing model was identified by comparing the AUC scores of the 3 models.</p></sec><sec id="s2-5-3"><title>Experiment 2: Across Cohorts</title><p>This experiment aims to address which data processing decisions most strongly impact model performance. For the purposes of analysis, the experiment is divided into 2 subsets: Set 1, which includes 16 cohorts derived from 4 key decisions (A, B, C, and D), and Set 2, which expands the design to 64 cohorts by incorporating 2 additional decisions, E and F. Each decision was implemented during both the training and testing phases. To clearly distinguish between the phases, a prime notation (eg, A&#x2032;, B&#x2032;, ..., F&#x2032;) is used to represent decisions applied during testing. To determine the influence of these data processing decisions, models from each cohort were compared using the following 3 analyses.</p><sec id="s2-5-3-1"><title>Approach 1: Maximum AUC Classification</title><p>For each test cohort, the model&#x2014;LR, RF, or GB&#x2014;that achieves the highest AUC is identified. A new binary column, &#x201C;max AUC,&#x201D; is added, where a value of 1 indicates that the model achieved the highest AUC for that cohort, and 0 otherwise. A &#x201C;DecisionTreeClassifier&#x201D; is then trained to determine which data cohort and decisions are most strongly associated with a model achieving the top AUC performance.</p></sec><sec id="s2-5-3-2"><title>Approach 2: Direct AUC Regression</title><p>This approach leverages the actual AUC values obtained by each model for every test cohort. A &#x201C;DecisionTreeRegressor&#x201D; is applied to examine how AUC values change across cohorts and to identify which data processing decisions most significantly influence model performance, either positively or negatively.</p></sec><sec id="s2-5-3-3"><title>Approach 3: AUC Gap Analysis</title><p>For each test cohort, the highest AUC achieved among the 3 models is identified. The AUC gap is then calculated as the difference between this maximum AUC and the AUC of each individual model. A &#x201C;DecisionTreeRegressor&#x201D; is used to analyze these gaps and determine which data processing decisions are most associated with performance loss relative to the best-performing model. These approaches provide complementary insights into how data processing decisions affect model accuracy and consistency across varying cohort definitions.</p></sec></sec></sec><sec id="s2-6"><title>Experiment 3: Across Demographic Groups</title><p>This experiment identifies how data processing decisions can impact model performance across demographic subgroups. Similar to Experiment 2, the same 3 analytic approaches&#x2014;maximum AUC classification, direct AUC regression, and AUC gap analysis&#x2014;are applied across the 3 models&#x2014;LR, RF, and GB.</p><p>However, in this experiment, the analysis is performed within stratified demographic groups. The test data are stratified by sex (male and female), race (Black, White, and Asian), and ethnicity (Hispanic or non-Hispanic). For each subgroup, shifts in model performance due to different data processing decisions are examined. This helps to determine whether certain decisions disproportionately impact certain demographic groups. The experiment is designed to help identify potential sources of bias and guides efforts toward more equitable predictive modeling.</p></sec><sec id="s2-7"><title>Ethical Considerations</title><p>The study was reviewed by the George Mason University Institutional Review Board (IRB), which determined that the proposed activity is not research involving human subjects as defined by the US Department of Health and Human Services (DHHS) and Food and Drug Administration (FDA) regulations. Therefore, the need for ethics approval was waived. The IRB reference number is #1706572. This study complies with the principles of the Declaration of Helsinki. The need for informed consent to participate was waived by the George Mason University IRB, as the study does not involve research on human subjects according to the relevant regulations.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>Consistent with the analytical framework outlined in the Methods, results are organized into 3 components reflecting key aspects of the study design. We first assess predictive performance across cohorts and demographic subpopulations to evaluate model accuracy and fairness. We then investigate the impact of data processing decisions on model performance across alternative cohort constructions. Finally, we examine the extent to which these effects vary across demographic subgroups. These findings collectively highlight the role of cohort selection and preprocessing decisions in shaping both predictive performance and equity.</p></sec><sec id="s3-2"><title>Model Performance Across Cohorts</title><p>Predictive performance across cohorts and demographic subpopulations demonstrated notable variability, reflecting the impact of cohort construction and data processing decisions on model outcomes. Across both cohort sets (Set 1 and Set 2), the models achieved moderate to high discriminative performance, with AUC values generally ranging from approximately 0.59 to 0.86 across different models and evaluation settings. However, performance was not consistent across demographic groups, with some subpopulations exhibiting wider variability in AUC ranges. These findings highlight the sensitivity of model performance to both cohort definition and population characteristics, underscoring the importance of evaluating predictive models across diverse groups.</p><p>It is important to note that for Native Hawaiian and multirace subgroups, some models show an AUC of zero due to the absence of data, leaving no cases for training or evaluation. <xref ref-type="table" rid="table1">Table 1</xref> summarizes the AUC ranges for LR, RF, and GB models across Cohorts 1&#x2010;16 and 1&#x2010;64, based on the entire dataset and stratified by demographic groups.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Range of area under the receiver-operator curve (AUC) values for logistic regression (LR), random forest (RF), and gradient boosting (GB) models across Cohorts 1&#x2010;16 and 1&#x2010;64, calculated using the entire dataset and stratified by demographic groups.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Cohort set</td><td align="left" valign="bottom">Entire datasets<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="bottom" colspan="7">Demographic group</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="top" colspan="2">Sex</td><td align="left" valign="top" colspan="3">Race</td><td align="left" valign="top" colspan="2">Ethnicity</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="top">Male</td><td align="left" valign="top">Female</td><td align="left" valign="top">Black</td><td align="left" valign="top">White</td><td align="left" valign="top">Asian</td><td align="left" valign="top">Hispanic</td><td align="left" valign="top">Non-Hispanic</td></tr></thead><tbody><tr><td align="left" valign="top">AUC<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> range (LR<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>, RF<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>, and GB<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup> models)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cohorts 1-16</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.70&#x2010;0.77<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></p></list-item><list-item><p>0.70&#x2010;0.84</p></list-item><list-item><p>0.63&#x2010;0.86</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.70&#x2010;0.74</p></list-item><list-item><p>0.69&#x2010;0.75</p></list-item><list-item><p>0.65&#x2010;0.75</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.70&#x2010;0.76</p></list-item><list-item><p>0.70&#x2010;0.76</p></list-item><list-item><p>0.66&#x2010;0.77</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.68&#x2010;0.74</p></list-item><list-item><p>0.68&#x2010;0.75</p></list-item><list-item><p>0.63&#x2010;0.75</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.70&#x2010;0.73</p></list-item><list-item><p>0.68&#x2010;0.75</p></list-item><list-item><p>0.64&#x2010;0.75</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.74&#x2010;0.80</p></list-item><list-item><p>0.70&#x2010;0.84</p></list-item><list-item><p>0.65&#x2010;0.82</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.70&#x2010;0.77</p></list-item><list-item><p>0.69&#x2010;0.77</p></list-item><list-item><p>0.67&#x2010;0.78</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.70&#x2010;0.74</p></list-item><list-item><p>0.68&#x2010;0.75</p></list-item><list-item><p>0.64&#x2010;0.75</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cohorts 1-64</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.69&#x2010;0.79</p></list-item><list-item><p>0.66&#x2010;0.78</p></list-item><list-item><p>0.59&#x2010;0.78</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.68&#x2010;0.78</p></list-item><list-item><p>0.67&#x2010;0.85</p></list-item><list-item><p>0.59&#x2010;0.90</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.70&#x2010;0.80</p></list-item><list-item><p>0.69&#x2010;0.87</p></list-item><list-item><p>0.60&#x2010;0.91</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.69&#x2010;0.78</p></list-item><list-item><p>0.68&#x2010;0.86</p></list-item><list-item><p>0.61&#x2010;0.90</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.68&#x2010;0.75</p></list-item><list-item><p>0.66&#x2010;0.82</p></list-item><list-item><p>0.57&#x2010;0.86</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.71&#x2010;0.81</p></list-item><list-item><p>0.66&#x2010;0.93</p></list-item><list-item><p>0.50&#x2010;0.97</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.71&#x2010;0.81</p></list-item><list-item><p>0.70&#x2010;0.91</p></list-item><list-item><p>0.62&#x2010;0.93</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>0.68&#x2010;0.77</p></list-item><list-item><p>0.66&#x2010;0.83</p></list-item><list-item><p>0.59&#x2010;0.88</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>The sequence of numbers in each cell, listed from top to bottom, represents the range of AUC values for the LR, RF, and GB models, respectively.</p></fn><fn id="table1fn2"><p><sup>b</sup>AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table1fn3"><p><sup>c</sup>LR: logistic regression.</p></fn><fn id="table1fn4"><p><sup>d</sup>RF: random forest.</p></fn><fn id="table1fn5"><p><sup>e</sup>GB: gradient boosting.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Influence of Data Processing Decisions</title><p>Model performance varied substantially across cohort definitions, indicating that data processing decisions play a critical role in shaping predictive outcomes. The importance of specific decisions differed depending on the cohort set and modeling context, suggesting that no single configuration consistently yields optimal performance. To better understand these effects, we further analyze how individual data processing decisions influence model accuracy across multiple evaluation settings.</p><p>To further interpret the findings, results are compared across multiple dimensions: within a single cohort set, across different cohort sets, across different models, and across the analytical approaches themselves. Due to page limitations, results are presented for a subset of approaches as representative examples.</p><sec id="s3-3-1"><title>Comparison 1: Within Cohort Set 1 (Cohorts 1-16)</title><p>This comparison focuses on the impact of the initial 4 data processing decisions&#x2014;A, B, C, and D&#x2014;across 16 distinct cohorts. The objective is to assess how variations in these decisions affect model performance when applied in both the training and testing sets. Due to space limitations, only a few illustrative plots are shown in this paper. For instance, feature importance of Approaches 1, 2, and 3 using the RF model comparing cohorts 1-16 for 4 main decisions is shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Feature importance for approaches 1, 2, and 3 using random forest (RF) model across cohorts 1-16 for 4 main decisions. RF: random forest. A, B, C, and D correspond to decisions applied to the training data, while A&#x2032;, B&#x2032;, C&#x2032;, and D&#x2032; correspond to decisions applied to the testing data.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88406_fig02.png"/></fig><p>Based on these results, decision D is concluded to play a crucial role across all 3 approaches. Specifically, including or excluding time-related information has a noticeable impact on model accuracy. Moreover, when considering A=A&#x2032;, B=B&#x2032;, C=C&#x2032;, and D=D&#x2032;, a more detailed understanding is gained. This alignment between decisions made during training (A, B, C, and D) and testing (A&#x2032;, B&#x2032;, C&#x2032;, and D&#x2032;) ensures that D=D&#x2032; remains consistent.</p><p>As shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>, the feature importance of D=D&#x2032; is significant across all approaches, highlighting its stability throughout both training and testing phases. Taken together, the results demonstrate that decision D is important and consistently influences model performance.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Feature importance for approaches 1, 2, and 3 using random forest (RF) model across cohorts 1-16 for 4 main decisions, offering deeper insights. RF: random forest.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88406_fig03.png"/></fig></sec><sec id="s3-3-2"><title>Comparison 2: Between Cohort Set 1 and Cohort Set 2</title><p>When comparing Set 1 (Cohorts 1-16) with Set 2 (Cohorts 1-64), decision D (time) remains highly important in Set 1. However, in Set 2, although decision D is not the most dominant factor, it still holds significant importance. This suggests that additional decisions&#x2014;such as F (Provider IDs)&#x2014;also play a crucial role in influencing model performance. Due to page limitations, only selected feature importance results comparing Set 1 (cohorts 1-16) and Set 2 (cohorts 1-64) using the LR model are shown in <xref ref-type="fig" rid="figure4">Figure 4</xref>.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Feature importance comparing Set 1 (cohorts 1 to 16) and Set 2 (cohorts 1 to 64) using the logistic regression (LR) model. LR: logistic regression.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88406_fig04.png"/></fig></sec><sec id="s3-3-3"><title>Comparison 3: Across 3 Different Models</title><p>This comparison examines how different ML models respond to the same data processing decisions. By analyzing 3 models, the study aims to understand whether the perceived importance of decisions remains stable across modeling approaches. The results reveal that the importance of decisions is not consistent across models. Each model assigns different levels of importance to the same decisions. Due to page limitations, only selected feature importance results for Approach 1 across the LR, RF, and GB models are shown in <xref ref-type="fig" rid="figure5">Figure 5</xref>.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Feature importance for Approach 1 compared across logistic regression (LR), random forest (RF), and gradient boosting (GB) models. GB: gradient boosting; LR: logistic regression; RF: random forest.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88406_fig05.png"/></fig></sec><sec id="s3-3-4"><title>Comparison 4: Among Different Analytical Approaches</title><p>This comparison assesses how the importance of data processing decisions varies across different analytical approaches. Decision D (time) consistently holds significant importance across all 3 approaches. However, the measurement ranges differ, indicating that changing the measurement scale can lead to different results. Due to page limitations, only selected feature importance comparisons of Approaches 1, 2, and 3 using the GB model are presented in <xref ref-type="fig" rid="figure6">Figure 6</xref>.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Feature importance comparing Approaches 1, 2, and 3 using the gradient boosting (GB) model. GB: gradient boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88406_fig06.png"/></fig></sec></sec><sec id="s3-4"><title>Differential Effects Across Demographic Subgroups</title><p>Model performance varied across demographic subgroups, indicating that the impact of data processing decisions is not uniform across populations. The importance of specific decisions differed by subgroup, suggesting potential disparities in model behavior and performance.</p><p>The impact of different decisions on cohort selection and cohort formation can be observed. These decisions affect subpopulations differently. Across different demographic groups, the quality of the model is influenced by different decisions to varying degrees. Certain decisions become more important depending on the subgroup, and the range of influential decisions varies across demographics. For illustration, a subset of representative plots is presented. Sex-specific differences (female and male) are provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>, while <xref ref-type="fig" rid="figure7">Figure 7</xref> presents differences by race (White and Asian) and <xref ref-type="fig" rid="figure8">Figure 8</xref> presents differences by ethnicity (Hispanic and non-Hispanic).</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Feature importance comparing race groups White and Asian for Approach 2 using the logistic regression (LR) model. LR: logistic regression.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88406_fig07.png"/></fig><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Feature importance comparing ethnicity groups Hispanic and non-Hispanic for Approach 3 using the logistic regression (LR) model. LR: logistic regression.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88406_fig08.png"/></fig><p>Fairness analysis using demographic parity and equality of opportunity revealed substantial variation across cohorts and demographic subgroups [<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref48">48</xref>]. While sex groups (female and male) exhibited relatively similar fairness patterns, notable disparities were observed across racial, ethnic, and age groups. In particular, certain racial categories and age groups showed consistently lower fairness metric values across multiple cohorts, indicating potential inequities in model performance (Figures S2 and S3 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>).</p><p>Because subgroup-level fairness metrics depend on both subgroup size and the distribution of outcome cases within each subgroup, these results should be interpreted cautiously. In this study, demographic parity was calculated as the proportion of predicted positive cases within each subgroup, while equality of opportunity and equalized odds were calculated among observed positive and negative cases, respectively. Therefore, when a subgroup contained very few patients, or when positive or negative outcome cases were absent, the corresponding fairness estimates could be unstable or not estimable. For this reason, fairness results for small demographic groups, including Native Hawaiian and multirace patients in some cohorts, should be interpreted as descriptive indicators of potential subgroup differences rather than definitive evidence of bias.</p><p>Calibration analysis further demonstrated variability in the alignment between predicted probabilities and observed outcomes. Across both Set 1 and Set 2, calibration curves showed that model reliability differs depending on cohort construction and training-testing combinations (Figures S4 and S5 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). These findings highlight that models with similar AUC values may exhibit different levels of calibration and fairness, reinforcing the importance of multi-dimensional evaluation. Due to space limitations, representative calibration plots for Set 1 and Set 2 are presented in Figures S4 and S5 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>, while the complete set of calibration results is provided in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendices 6</xref> and <xref ref-type="supplementary-material" rid="app7">7</xref>.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>The study systematically explores the impact of cohort selection criteria on ML model performance and demographic disparities in predicting COVID-19 outcomes. Overall, the findings address the study objective by showing that model performance and subgroup results are sensitive to cohort construction decisions. The findings underscore the critical role of data processing decisions in shaping model accuracy. Notably, decision D (time inclusion/exclusion) was consistently associated with variation in model performance, particularly in Set 1, which comprises 16 cohorts based on 4 primary decisions.</p><p>The analysis reveals that different ML models&#x2014;LR, RF, and GB&#x2014;respond variably to the same data processing decisions, indicating that the perceived importance of these decisions is not consistent across models. This variability highlights the need for careful consideration of model choice in conjunction with cohort selection criteria to ensure robust and reliable predictions. These findings reinforce prior concerns in the literature that differences in data preprocessing&#x2014;rather than model selection alone&#x2014;can substantially influence observed performance, thereby affecting reproducibility and generalizability.</p><p>Furthermore, the study delves into the impact of data processing decisions on model performance across demographic subgroups, including gender, race, and ethnicity. The results indicate that certain decisions disproportionately affect specific demographic groups, potentially introducing bias and exacerbating health disparities. This finding emphasizes the importance of fair and transparent cohort selection practices to promote equity in ML-based health care research. Importantly, these results suggest that fairness is not solely a property of the algorithm but is strongly influenced by upstream data construction choices.</p><p>Although the AUC is widely used to evaluate model discrimination, it does not capture several critical dimensions of model performance, including calibration, subgroup-specific error patterns, and fairness across populations. As a result, models with similar AUC values may differ substantially in terms of reliability and equity. To address this limitation, this study incorporates additional evaluation dimensions, including fairness metrics and calibration analyses, to provide a more comprehensive assessment of model performance.</p><p>Furthermore, although confidence intervals and formal statistical tests are important for quantifying uncertainty, the large sample size of the N3C dataset inherently provides high statistical power to detect differences in AUC, as demonstrated in prior methodological studies. However, due to the computational complexity associated with large-scale data, resampling-based approaches such as bootstrapping were not feasible within the study environment [<xref ref-type="bibr" rid="ref49">49</xref>]. Future work should explore scalable and computationally efficient methods for uncertainty estimation in large clinical datasets. So, these findings underscore that fairness assessment should not rely solely on discrimination metrics such as AUC, but instead should integrate multiple complementary measures to ensure a robust, reliable, and equitable evaluation of model performance.</p><p>Threshold-dependent metrics, such as false positive and false negative rate parity, provide important insights into subgroup-specific error disparities but require the selection of operating thresholds, which may vary across settings. In this study, we focused on threshold-independent metrics and calibration for consistency across cohorts. Future work will extend this analysis to include threshold-dependent fairness measures.</p><p>Importantly, this study was designed to evaluate how alternative cohort construction decisions affect model performance and subgroup estimates, rather than to identify a single optimal cohort definition. Therefore, the findings should be interpreted as demonstrating sensitivity to cohort construction choices, not as establishing a validated guideline for fair or standardized cohort selection. Development of such guidelines would require additional clinical validation, external replication, and evaluation across multiple datasets, outcomes, and health care settings.</p><p>The study also highlights the challenges posed by inconsistencies in cohort construction practices and the absence of standardized guidelines, which can hinder the comparability and validation of ML models across different settings. Addressing these challenges requires a structured and transparent approach to cohort definition, accounting for equity, clinical relevance, and methodological rigor. From a broader perspective, these findings suggest the need for future work to develop and validate standardized cohort design frameworks that may improve consistency across studies and support more reliable cross-study comparisons.</p><p>This research underscores the importance of considering cohort definition and data processing decisions when designing clinical ML models, especially in the context of equity and performance consistency. In addition to these contributions, several limitations should be considered when interpreting the findings. The study shows that cohort selection criteria can influence ML model performance and demographic subgroup results in clinical outcome prediction. Decision D (time inclusion/exclusion) was consistently associated with variation in model performance, particularly in Set 1, while other decisions, such as provider filtering, also contributed to performance differences in Set 2. The findings suggest that certain data processing decisions may affect demographic groups differently, highlighting the need for transparent reporting and sensitivity analysis in cohort construction. By systematically analyzing combinations of data processing criteria across multiple modeling approaches, this work provides insight into the potential consequences of cohort definition choices. This study relied on traditional ML methods; future work should examine whether similar patterns are observed with deep learning&#x2013;based approaches.</p><p>Feature importance derived from tree-based models may be sensitive to data structure and model initialization and may exhibit variability under different sampling or training conditions. In addition, such measures can be influenced by correlated or structured features, which may introduce bias in the relative importance rankings. Therefore, the reported importance values should be interpreted with caution and should not be considered as evidence of causal relationships. Although consistent patterns were observed across multiple models and analytical approaches in this study, future work should further assess the stability and robustness of these findings using repeated resampling strategies, multiple random seeds, and complementary importance measures such as permutation importance or Shapley Additive Explanations (SHAP) values.</p><p>Some demographic subgroups, particularly Native Hawaiian and multirace, had very small or zero sample sizes in certain cohorts. In several cases, these groups were absent from either the training or testing sets, making it impossible to compute reliable performance metrics such as AUC. Consequently, reported values of zero or undefined AUC reflect data limitations rather than true model performance. Small sample sizes increase variability and reduce the statistical reliability of subgroup-level estimates, which may lead to unstable or misleading comparisons. Therefore, results for underrepresented subgroups should be interpreted with caution, and fairness assessments are more reliable for groups with sufficient representation. This limitation highlights the importance of adequate subgroup sample sizes when evaluating model performance across demographic populations.</p><p>Subgroup fairness analyses were affected by small sample sizes and limited outcome variation in some demographic groups. Although <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> reports cohort-level sample sizes, train/test splits, and demographic distributions across the 16- and 64-cohort configurations, some subgroups, particularly multirace and Native Hawaiian patients, had very small counts in several cohorts. Because subgroup AUC requires both positive and negative outcome cases, and threshold-based fairness metrics depend on the distribution of observed outcomes and predicted classifications within each subgroup, estimates for sparsely represented groups may be unstable. Therefore, subgroup fairness results for small demographic groups should be interpreted cautiously as descriptive indicators rather than definitive evidence of model fairness or unfairness. Future studies should build on these descriptive summaries by reporting subgroup-specific outcome counts and uncertainty measures alongside fairness metrics.</p><p>This study has several limitations that should be acknowledged. First, use of data from a specific time period (August 1, 2020-December 31, 2021) may not capture the changing nature of COVID-19, including changes in virus variant, vaccination rates, and health care practices. Therefore, our findings may not be generalizable to current or future scenarios. Second, the study is limited to adult patients (18 years and older), thereby excluding pediatric populations. Accordingly, the results cannot be generalized to younger populations, and differences in the presentation and outcome of COVID-19 among children and adolescents were not examined.</p><p>Although 18 years is widely used as the standard threshold for defining adult populations in clinical research and public health reporting, alternative age thresholds could result in modest differences in cohort composition. Future studies could evaluate the sensitivity of the findings to alternative age cutoffs to further assess the robustness of the observed results.</p><p>Finally, while the study simplifies data processing decisions to maintain analytic feasibility, limiting the study to only including other possibly influential factors such as socioeconomic status, comorbidities, or geographic variations would give a more complete picture of model performance and demographic disparities.</p></sec><sec id="s4-2"><title>Conclusions</title><p>In conclusion, this study demonstrates that cohort selection is not merely a preprocessing step but a fundamental driver of ML performance and fairness in clinical research. Overall, the study calls for the development of comprehensive guidelines for cohort construction and the validation of existing models against diverse benchmarks to ensure ML models are valid, equitable, and reproducible. These efforts are essential for achieving trustworthy and impactful ML applications in health care, particularly in the context of the COVID-19 pandemic and beyond. More broadly, these findings highlight that improving equity and reliability in clinical ML requires careful attention to upstream data decisions, emphasizing transparency, standardization, and reproducibility as core principles for future research.</p></sec></sec></body><back><ack><p>This study was conducted as part of the National COVID Cohort Collaborative (N3C) Consortium. Authorship was determined using ICMJE recommendations. The analyses described in this publication were conducted with data and tools accessed through the NCATS N3C Data Enclave in accordance with the N3C Attribution &#x0026; Publication Policy v1.2-2020-08-25b. This work was supported by NCATS Contract No. 75N95023D00001, Axle Informatics Subcontract: NCATS-P00438-B, and CD2H &#x2013; The National COVID Cohort Collaborative (N3C) IDeA CTR Collaboration (3U24TR002306-04S2; NCATS U24 TR002306). This research was made possible by the patients whose data are included in the N3C Data Enclave, as well as the contributing organizations, signatories, and scientists who have supported the ongoing development of this community resource. Additional details about N3C are available in the foundational publication by Haendel et al [<xref ref-type="bibr" rid="ref50">50</xref>] and related documentation.</p><p>The authors used a generative AI tool (ChatGPT) in a limited capacity solely for language editing and grammar correction. All scientific content, study design, analyses, and interpretations were developed and verified by the authors.</p></ack><notes><sec><title>Funding</title><p>No funding was received for conducting this study.</p></sec><sec><title>Data Availability</title><p>The datasets analyzed during this study are not publicly available because they contain restricted patient-level electronic health record data. The data may be accessed by qualified researchers through the NCATS N3C Data Enclave, subject to N3C access requirements, institutional approval, completion of the required training, and applicable data use agreements. The analyses were conducted in accordance with the N3C Attribution &#x0026; Publication Policy v1.2-2020-08-25b.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb2">BP</term><def><p>blood pressure</p></def></def-item><def-item><term id="abb3">CBC</term><def><p>complete blood count</p></def></def-item><def-item><term id="abb4">CDM</term><def><p>Common Data Model</p></def></def-item><def-item><term id="abb5">DHHS</term><def><p>Department of Health and Human Services</p></def></def-item><def-item><term id="abb6">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb7">FDA</term><def><p>Food and Drug Administration</p></def></def-item><def-item><term id="abb8">GB</term><def><p>gradient boosting</p></def></def-item><def-item><term id="abb9">GFR</term><def><p>glomerular filtration rate</p></def></def-item><def-item><term id="abb10">IRB</term><def><p>Institutional Review Board</p></def></def-item><def-item><term id="abb11">LDS</term><def><p>limited dataset</p></def></def-item><def-item><term id="abb12">LR</term><def><p>logistic regression</p></def></def-item><def-item><term id="abb13">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb14">N3C</term><def><p>National COVID Cohort Collaborative</p></def></def-item><def-item><term id="abb15">NT-proBNP</term><def><p>N-terminal pro&#x2013;B-type natriuretic peptide</p></def></def-item><def-item><term id="abb16">OMOP</term><def><p>Observational Medical Outcomes Partnership</p></def></def-item><def-item><term id="abb17">RF</term><def><p>random forest</p></def></def-item><def-item><term id="abb18">SHAP</term><def><p>Shapley Additive Explanations</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rajkomar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dean</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kohane</surname><given-names>I</given-names> </name></person-group><article-title>Machine learning in medicine</article-title><source>N Engl J Med</source><year>2019</year><month>04</month><day>4</day><volume>380</volume><issue>14</issue><fpage>1347</fpage><lpage>1358</lpage><pub-id pub-id-type="doi">10.1056/NEJMra1814259</pub-id><pub-id pub-id-type="medline">30943338</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hong</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>The development and validation of postpartum hemorrhage prediction models for pregnancies with placenta previa totalis based on coagulation function indexes: a retrospective cohort study</article-title><source>BMC Pregnancy Childbirth</source><year>2025</year><month>09</month><day>2</day><volume>25</volume><issue>1</issue><fpage>925</fpage><pub-id pub-id-type="doi">10.1186/s12884-025-08066-1</pub-id><pub-id pub-id-type="medline">40898059</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Purushotham</surname><given-names>S</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Che</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name></person-group><article-title>Benchmarking deep learning models on large healthcare datasets</article-title><source>J Biomed Inform</source><year>2018</year><month>07</month><volume>83</volume><fpage>112</fpage><lpage>134</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2018.04.007</pub-id><pub-id pub-id-type="medline">29879470</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Qui&#x00F1;onero-Candela</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sugiyama</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schwaighofer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lawrence</surname><given-names>ND</given-names> </name></person-group><source>Dataset Shift in Machine Learning</source><year>2008</year><publisher-name>MIT Press</publisher-name><pub-id pub-id-type="doi">10.7551/mitpress/9780262170055.001.0001</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Guerra-Manzanares</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lopez</surname><given-names>LJL</given-names> </name><name name-style="western"><surname>Maniatakos</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shamout</surname><given-names>FE</given-names> </name></person-group><article-title>Privacy-preserving machine learning for healthcare: open challenges and future perspectives</article-title><source>Trustworthy Machine Learning for Healthcare</source><year>2023</year><volume>13932</volume><fpage>25</fpage><lpage>40</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-39539-0_3</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Andaur Navarro</surname><given-names>CL</given-names> </name><name name-style="western"><surname>Damen</surname><given-names>JAA</given-names> </name><name name-style="western"><surname>Takada</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Risk of bias in studies on prediction models developed using supervised machine learning techniques: systematic review</article-title><source>BMJ</source><year>2021</year><month>10</month><day>20</day><volume>375</volume><fpage>n2281</fpage><pub-id pub-id-type="doi">10.1136/bmj.n2281</pub-id><pub-id pub-id-type="medline">34670780</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Obermeyer</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Powers</surname><given-names>B</given-names> </name><name name-style="western"><surname>Vogeli</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mullainathan</surname><given-names>S</given-names> </name></person-group><article-title>Dissecting racial bias in an algorithm used to manage the health of populations</article-title><source>Science</source><year>2019</year><month>10</month><day>25</day><volume>366</volume><issue>6464</issue><fpage>447</fpage><lpage>453</lpage><pub-id pub-id-type="doi">10.1126/science.aax2342</pub-id><pub-id pub-id-type="medline">31649194</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cross</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Choma</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Onofrey</surname><given-names>JA</given-names> </name></person-group><article-title>Bias in medical AI: implications for clinical decision-making</article-title><source>PLoS Digit Health</source><year>2024</year><month>11</month><volume>3</volume><issue>11</issue><fpage>e0000651</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000651</pub-id><pub-id pub-id-type="medline">39509461</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Banda</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Sarraju</surname><given-names>A</given-names> </name><name name-style="western"><surname>Abbasi</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Finding missed cases of familial hypercholesterolemia in health systems using machine learning</article-title><source>NPJ Digit Med</source><year>2019</year><volume>2</volume><issue>1</issue><fpage>23</fpage><pub-id pub-id-type="doi">10.1038/s41746-019-0101-5</pub-id><pub-id pub-id-type="medline">31304370</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gong</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Su</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shu</surname><given-names>J</given-names> </name></person-group><article-title>Prediction of angiogenesis in extrahepatic cholangiocarcinoma using MRI-based machine learning</article-title><source>Front Oncol</source><year>2023</year><volume>13</volume><fpage>1048311</fpage><pub-id pub-id-type="doi">10.3389/fonc.2023.1048311</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koola</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Ho</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Development of a national department of Veterans Affairs mortality risk prediction model among patients with cirrhosis</article-title><source>BMJ Open Gastroenterol</source><year>2019</year><volume>6</volume><issue>1</issue><fpage>e000342</fpage><pub-id pub-id-type="doi">10.1136/bmjgast-2019-000342</pub-id><pub-id pub-id-type="medline">31875140</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mehrabi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Morstatter</surname><given-names>F</given-names> </name><name name-style="western"><surname>Saxena</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lerman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Galstyan</surname><given-names>A</given-names> </name></person-group><article-title>A survey on bias and fairness in machine learning</article-title><source>ACM Comput Surv</source><year>2022</year><month>07</month><day>31</day><volume>54</volume><issue>6</issue><fpage>1</fpage><lpage>35</lpage><pub-id pub-id-type="doi">10.1145/3457607</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pool</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hebdon</surname><given-names>M</given-names> </name><name name-style="western"><surname>de Groot</surname><given-names>E</given-names> </name><etal/></person-group><article-title>A novel approach for assessing bias during team-based clinical decision-making</article-title><source>Front Public Health</source><year>2023</year><volume>11</volume><fpage>1014773</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2023.1014773</pub-id><pub-id pub-id-type="medline">37228737</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Breathett</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lum</surname><given-names>HD</given-names> </name><etal/></person-group><article-title>Factors related to physician clinical decision-making for African-American and Hispanic patients: a qualitative meta-synthesis</article-title><source>J Racial Ethn Health Disparities</source><year>2018</year><month>12</month><volume>5</volume><issue>6</issue><fpage>1215</fpage><lpage>1229</lpage><pub-id pub-id-type="doi">10.1007/s40615-018-0468-z</pub-id><pub-id pub-id-type="medline">29508374</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Breathett</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yee</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pool</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Association of gender and race with allocation of advanced heart failure therapies</article-title><source>JAMA Netw Open</source><year>2020</year><month>07</month><day>1</day><volume>3</volume><issue>7</issue><fpage>e2011044</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2020.11044</pub-id><pub-id pub-id-type="medline">32692370</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Innes</surname><given-names>H</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>P</given-names> </name><name name-style="western"><surname>McDonald</surname><given-names>SA</given-names> </name><etal/></person-group><article-title>Competing risk bias in prognostic models predicting hepatocellular carcinoma occurrence: impact on clinical decision-making</article-title><source>Gastro Hep Adv</source><year>2022</year><volume>1</volume><issue>2</issue><fpage>129</fpage><lpage>136</lpage><pub-id pub-id-type="doi">10.1016/j.gastha.2021.11.008</pub-id><pub-id pub-id-type="medline">39131124</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haghighathoseini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wojtusiak</surname><given-names>J</given-names> </name><name name-style="western"><surname>Min</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Does cohort selection affect machine learning from clinical data?</article-title><source>AMIA Annu Symp Proc</source><year>2024</year><volume>2024</volume><fpage>473</fpage><lpage>482</lpage><pub-id pub-id-type="medline">40417536</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>O&#x2019;Sullivan</surname><given-names>ED</given-names> </name><name name-style="western"><surname>Schofield</surname><given-names>SJ</given-names> </name></person-group><article-title>Cognitive bias in clinical medicine</article-title><source>J R Coll Physicians Edinb</source><year>2018</year><month>09</month><volume>48</volume><issue>3</issue><fpage>225</fpage><lpage>232</lpage><pub-id pub-id-type="doi">10.4997/JRCPE.2018.306</pub-id><pub-id pub-id-type="medline">30191910</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haghighathoseini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wojtusiak</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ngana</surname><given-names>LP</given-names> </name><name name-style="western"><surname>Min</surname><given-names>H</given-names> </name><name name-style="western"><surname>Menon</surname><given-names>NM</given-names> </name></person-group><article-title>Which decisions affect cohort distribution in COVID-19 data analytics?</article-title><source>AMIA Annu Symp Proc</source><year>2024</year><volume>2024</volume><fpage>413</fpage><lpage>422</lpage><pub-id pub-id-type="medline">41726482</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yi</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Computed tomography radiomics for predicting pathological grade of renal cell carcinoma</article-title><source>Front Oncol</source><year>2020</year><volume>10</volume><fpage>570396</fpage><pub-id pub-id-type="doi">10.3389/fonc.2020.570396</pub-id><pub-id pub-id-type="medline">33585193</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Laukhtina</surname><given-names>E</given-names> </name><name name-style="western"><surname>Schuettfort</surname><given-names>VM</given-names> </name><name name-style="western"><surname>D&#x2019;Andrea</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Selection and evaluation of preoperative systemic inflammatory response biomarkers model prior to cytoreductive nephrectomy using a machine-learning approach</article-title><source>World J Urol</source><year>2022</year><month>03</month><volume>40</volume><issue>3</issue><fpage>747</fpage><lpage>754</lpage><pub-id pub-id-type="doi">10.1007/s00345-021-03844-w</pub-id><pub-id pub-id-type="medline">34671856</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Clichet</surname><given-names>V</given-names> </name><name name-style="western"><surname>Lebon</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chapuis</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Artificial intelligence to empower diagnosis of myelodysplastic syndromes by multiparametric flow cytometry</article-title><source>Haematologica</source><year>2023</year><month>09</month><day>1</day><volume>108</volume><issue>9</issue><fpage>2435</fpage><lpage>2443</lpage><pub-id pub-id-type="doi">10.3324/haematol.2022.282370</pub-id><pub-id pub-id-type="medline">36924240</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chaganti</surname><given-names>S</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>V</given-names> </name><name name-style="western"><surname>Gent</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Kamaleswaran</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kamen</surname><given-names>A</given-names> </name></person-group><article-title>Evaluating the impact of common clinical confounders on performance of deep-learning based sepsis risk assessment</article-title><source>In Review</source><comment>Preprint posted online on  Apr 10, 2024</comment><pub-id pub-id-type="doi">10.21203/rs.3.rs-4017967/v1</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>El-Rashidy</surname><given-names>N</given-names> </name><name name-style="western"><surname>Abdelrazik</surname><given-names>S</given-names> </name><name name-style="western"><surname>Abuhmed</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Comprehensive survey of using machine learning in the COVID-19 pandemic</article-title><source>Diagnostics (Basel)</source><year>2021</year><month>06</month><day>24</day><volume>11</volume><issue>7</issue><fpage>1155</fpage><pub-id pub-id-type="doi">10.3390/diagnostics11071155</pub-id><pub-id pub-id-type="medline">34202587</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wynants</surname><given-names>L</given-names> </name><name name-style="western"><surname>Van Calster</surname><given-names>B</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>GS</given-names> </name><etal/></person-group><article-title>Prediction models for diagnosis and prognosis of COVID-19: systematic review and critical appraisal</article-title><source>BMJ</source><year>2020</year><month>04</month><day>7</day><volume>369</volume><fpage>m1328</fpage><pub-id pub-id-type="doi">10.1136/bmj.m1328</pub-id><pub-id pub-id-type="medline">32265220</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tay</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yen</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Rivera</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Development and external validation of clinical features-based machine learning models for predicting COVID-19 in the emergency department</article-title><source>West J Emerg Med</source><year>2024</year><month>01</month><volume>25</volume><issue>1</issue><fpage>67</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.5811/westjem.60243</pub-id><pub-id pub-id-type="medline">38205987</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Haghighathoseini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wojtusiak</surname><given-names>J</given-names> </name><name name-style="western"><surname>Menon</surname><given-names>NM</given-names> </name><name name-style="western"><surname>Min</surname><given-names>H</given-names> </name><name name-style="western"><surname>Frankenfeld</surname><given-names>C</given-names> </name><name name-style="western"><surname>Leslie</surname><given-names>T</given-names> </name></person-group><article-title>Big data decision-making and racial disparities: a case study among COVID-19 inpatient visits</article-title><conf-name>2024 IEEE International Conference on Big Data (BigData)</conf-name><conf-date>Dec 15-18, 2024</conf-date><conf-loc>Washington, DC</conf-loc><fpage>6452</fpage><lpage>6459</lpage><pub-id pub-id-type="doi">10.1109/BigData62323.2024.10825471</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moneim</surname><given-names>IA</given-names> </name><name name-style="western"><surname>El-Latif</surname><given-names>EIA</given-names> </name></person-group><article-title>Modelling the fourth wave of COVID-19 pandemic in Egypt</article-title><source>J Math Computer Sci</source><year>2023</year><volume>29</volume><issue>1</issue><fpage>52</fpage><lpage>59</lpage><pub-id pub-id-type="doi">10.22436/jmcs.029.01.05</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Haghighathoseini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Qodrati</surname><given-names>M</given-names> </name><name name-style="western"><surname>Min</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Selection Bias from Data Processing in N3C</article-title><conf-name>2024 IEEE 12th International Conference on Healthcare Informatics (ICHI)</conf-name><conf-date>Jun 3-6, 2024</conf-date><conf-loc>Orlando, FL</conf-loc><fpage>234</fpage><lpage>241</lpage><pub-id pub-id-type="doi">10.1109/ICHI61247.2024.00038</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>A novel explainable online calculator for contrast-induced AKI in diabetics: a multi-centre validation and prospective evaluation study</article-title><source>J Transl Med</source><year>2023</year><month>07</month><day>31</day><volume>21</volume><issue>1</issue><fpage>517</fpage><pub-id pub-id-type="doi">10.1186/s12967-023-04387-x</pub-id><pub-id pub-id-type="medline">37525240</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Development of a COVID-19 early risk assessment system based on multiple machine learning algorithms and routine blood tests: a real-world study</article-title><source>Front Immunol</source><year>2024</year><volume>15</volume><fpage>1430899</fpage><pub-id pub-id-type="doi">10.3389/fimmu.2024.1430899</pub-id><pub-id pub-id-type="medline">39403385</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>K</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Frangi</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Schnabel</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Davatzikos</surname><given-names>C</given-names> </name><name name-style="western"><surname>Alberola-L&#x00F3;pez</surname><given-names>C</given-names> </name><name name-style="western"><surname>Fichtinger</surname><given-names>G</given-names> </name></person-group><article-title>Generative invertible networks (GIN): pathophysiology-interpretable feature mapping and virtual patient generation</article-title><year>2018</year><conf-name>Medical Image Computing and Computer Assisted Intervention &#x2013; MICCAI 2018</conf-name><conf-date>Sep 16-20, 2018</conf-date><conf-loc>Granada, Spain</conf-loc><publisher-name>Springer</publisher-name><fpage>537</fpage><lpage>545</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-00928-1_61</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Lei</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Machine learning predicts cancer-associated venous thromboembolism using clinically available variables in gastric cancer patients</article-title><source>Heliyon</source><year>2023</year><month>01</month><volume>9</volume><issue>1</issue><fpage>e12681</fpage><pub-id pub-id-type="doi">10.1016/j.heliyon.2022.e12681</pub-id><pub-id pub-id-type="medline">36632097</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Job</surname><given-names>C</given-names> </name><name name-style="western"><surname>Adenipekun</surname><given-names>B</given-names> </name><name name-style="western"><surname>Cleves</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gill</surname><given-names>P</given-names> </name><name name-style="western"><surname>Samuriwo</surname><given-names>R</given-names> </name></person-group><article-title>Health professionals implicit bias of patients with low socioeconomic status (SES) and its effects on clinical decision-making: a scoping review</article-title><source>BMJ Open</source><year>2024</year><month>07</month><day>2</day><volume>14</volume><issue>7</issue><fpage>e081723</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2023-081723</pub-id><pub-id pub-id-type="medline">38960454</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Banerjee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Friedman</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Klesse</surname><given-names>LJ</given-names> </name><etal/></person-group><article-title>COVID-19 in people with neurofibromatosis 1, neurofibromatosis 2, or schwannomatosis</article-title><source>Genet Med</source><year>2023</year><month>02</month><volume>25</volume><issue>2</issue><fpage>100324</fpage><pub-id pub-id-type="doi">10.1016/j.gim.2022.10.007</pub-id><pub-id pub-id-type="medline">36565307</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bennett</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Moffitt</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Hajagos</surname><given-names>JG</given-names> </name><etal/></person-group><article-title>Clinical characterization and prediction of clinical severity of SARS-CoV-2 infection among US adults using data from the US National COVID cohort collaborative</article-title><source>JAMA Netw Open</source><year>2021</year><month>07</month><day>1</day><volume>4</volume><issue>7</issue><fpage>e2116901</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2021.16901</pub-id><pub-id pub-id-type="medline">34255046</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="web"><article-title>National Clinical Cohort Collaborative (N3C) Homepage Enclave and Tenants</article-title><source>National Clinical Cohort Collaborative</source><access-date>2023-12-21</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://covid.cd2h.org/">https://covid.cd2h.org/</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jones</surname><given-names>M</given-names> </name><name name-style="western"><surname>Winger</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wernz</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Investigating the impact of temporal labeling of emergency department visits for COVID-19: comparing healthcare disparities analyses using comprehensive, single-site data with national COVID cohort collaborative (N3C) data</article-title><conf-name>2023 Systems and Information Engineering Design Symposium (SIEDS)</conf-name><conf-date>Apr 27-28, 2023</conf-date><conf-loc>Charlottesville, VA</conf-loc><fpage>297</fpage><lpage>302</lpage><pub-id pub-id-type="doi">10.1109/SIEDS58326.2023.10137801</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sharafeldin</surname><given-names>N</given-names> </name><name name-style="western"><surname>Bates</surname><given-names>B</given-names> </name><name name-style="western"><surname>Song</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Outcomes of COVID-19 in patients with cancer: report from the National COVID Cohort Collaborative (N3C)</article-title><source>J Clin Oncol</source><year>2021</year><month>07</month><day>10</day><volume>39</volume><issue>20</issue><fpage>2232</fpage><lpage>2246</lpage><pub-id pub-id-type="doi">10.1200/JCO.21.01074</pub-id><pub-id pub-id-type="medline">34085538</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leese</surname><given-names>P</given-names> </name><name name-style="western"><surname>Anand</surname><given-names>A</given-names> </name><name name-style="western"><surname>Girvin</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Clinical encounter heterogeneity and methods for resolving in networked EHR data: a study from N3C and RECOVER programs</article-title><source>J Am Med Inform Assoc</source><year>2023</year><month>05</month><day>19</day><volume>30</volume><issue>6</issue><fpage>1125</fpage><lpage>1136</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad057</pub-id><pub-id pub-id-type="medline">37087110</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wolpert</surname><given-names>DH</given-names> </name><name name-style="western"><surname>Macready</surname><given-names>WG</given-names> </name></person-group><article-title>No free lunch theorems for optimization</article-title><source>IEEE Trans Evol Computat</source><year>1997</year><month>04</month><volume>1</volume><issue>1</issue><fpage>67</fpage><lpage>82</lpage><pub-id pub-id-type="doi">10.1109/4235.585893</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ngana</surname><given-names>LP</given-names> </name><name name-style="western"><surname>Haghighathoseini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wojtusiak</surname><given-names>J</given-names> </name></person-group><article-title>Influence of stratified variable encoding on quality of mortality prediction in systolic heart failure</article-title><conf-name>2025 International Conference on Machine Learning and Applications (ICMLA)</conf-name><conf-date>Dec 3-5, 2025</conf-date><conf-loc>Boca Raton, FL</conf-loc><fpage>996</fpage><lpage>999</lpage><pub-id pub-id-type="doi">10.1109/ICMLA66185.2025.00150</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cha</surname><given-names>GW</given-names> </name><name name-style="western"><surname>Moon</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>YC</given-names> </name></person-group><article-title>Comparison of random forest and gradient boosting machine models for predicting demolition waste based on small datasets and categorical variables</article-title><source>Int J Environ Res Public Health</source><year>2021</year><month>08</month><day>12</day><volume>18</volume><issue>16</issue><fpage>8530</fpage><pub-id pub-id-type="doi">10.3390/ijerph18168530</pub-id><pub-id pub-id-type="medline">34444277</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nusinovici</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tham</surname><given-names>YC</given-names> </name><name name-style="western"><surname>Chak Yan</surname><given-names>MY</given-names> </name><etal/></person-group><article-title>Logistic regression was as good as machine learning for predicting major chronic diseases</article-title><source>J Clin Epidemiol</source><year>2020</year><month>06</month><volume>122</volume><fpage>56</fpage><lpage>69</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2020.03.002</pub-id><pub-id pub-id-type="medline">32169597</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Collins</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Reitsma</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>DG</given-names> </name><name name-style="western"><surname>Moons</surname><given-names>KGM</given-names> </name><collab>TRIPOD Group</collab></person-group><article-title>Transparent reporting of a multivariable prediction model for individual prognosis or diagnosis (TRIPOD): the TRIPOD statement. The TRIPOD Group</article-title><source>Circulation</source><year>2015</year><month>01</month><day>13</day><volume>131</volume><issue>2</issue><fpage>211</fpage><lpage>219</lpage><pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.114.014508</pub-id><pub-id pub-id-type="medline">25561516</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Richter</surname><given-names>LR</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Causal fairness assessment of treatment allocation with electronic health records</article-title><source>J Biomed Inform</source><year>2024</year><month>07</month><volume>155</volume><fpage>104656</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2024.104656</pub-id><pub-id pub-id-type="medline">38782170</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kecki</surname><given-names>V</given-names> </name><name name-style="western"><surname>Said</surname><given-names>A</given-names> </name></person-group><article-title>Understanding fairness in recommender systems: a healthcare perspective</article-title><conf-name>RecSys &#x2019;24: Proceedings of the 18th ACM Conference on Recommender Systems</conf-name><conf-date>Oct 14-18, 2024</conf-date><conf-loc>Bari, Italy</conf-loc><fpage>1125</fpage><lpage>1130</lpage><pub-id pub-id-type="doi">10.1145/3640457.3691711</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pfohl</surname><given-names>S</given-names> </name><name name-style="western"><surname>Marafino</surname><given-names>B</given-names> </name><name name-style="western"><surname>Coulet</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rodriguez</surname><given-names>F</given-names> </name><name name-style="western"><surname>Palaniappan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>NH</given-names> </name></person-group><article-title>Creating fair models of atherosclerotic cardiovascular disease risk</article-title><conf-name>AIES &#x2019;19: Proceedings of the 2019 AAAI/ACM Conference on AI, Ethics, and Society</conf-name><conf-date>Jan 27-28, 2019</conf-date><conf-loc>Honolulu HI</conf-loc><fpage>271</fpage><lpage>278</lpage><pub-id pub-id-type="doi">10.1145/3306618.3314278</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Emir</surname><given-names>B</given-names> </name><name name-style="western"><surname>Can</surname><given-names>FE</given-names> </name><name name-style="western"><surname>Kaymaz</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Sample size and power analysis for ROC AUC differences in diagnostic tests: a methodological evaluation of the Obuchowski-McClish and Hanley-McNeil methods</article-title><source>BMC Med Res Methodol</source><year>2026</year><month>01</month><day>28</day><volume>26</volume><issue>1</issue><fpage>43</fpage><pub-id pub-id-type="doi">10.1186/s12874-026-02768-6</pub-id><pub-id pub-id-type="medline">41606502</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haendel</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Chute</surname><given-names>CG</given-names> </name><name name-style="western"><surname>Bennett</surname><given-names>TD</given-names> </name><etal/></person-group><article-title>The National COVID Cohort Collaborative (N3C): rationale, design, infrastructure, and deployment</article-title><source>J Am Med Inform Assoc</source><year>2021</year><month>03</month><day>1</day><volume>28</volume><issue>3</issue><fpage>427</fpage><lpage>443</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaa196</pub-id><pub-id pub-id-type="medline">32805036</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Sample sizes and demographic characteristics across the 16- and 64-cohort configurations.</p><media xlink:href="formative_v10i1e88406_app1.docx" xlink:title="DOCX File, 96 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Performance metrics for logistic regression, random forest, and gradient boosting models across 16 cohorts, overall and by demographic subgroup.</p><media xlink:href="formative_v10i1e88406_app2.docx" xlink:title="DOCX File, 674 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Performance metrics for logistic regression, random forest, and gradient boosting models across 64 cohorts.</p><media xlink:href="formative_v10i1e88406_app3.docx" xlink:title="DOCX File, 1410 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Demographic subgroup-specific AUC values for logistic regression, random forest, and gradient boosting models across 64 cohorts.</p><media xlink:href="formative_v10i1e88406_app4.docx" xlink:title="DOCX File, 2641 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Supplementary analyses of feature importance, demographic fairness, and model calibration across the 16- and 64-cohort configurations.</p><media xlink:href="formative_v10i1e88406_app5.docx" xlink:title="DOCX File, 2862 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Calibration plots for logistic regression, random forest, and gradient boosting models across 16 cohorts.</p><media xlink:href="formative_v10i1e88406_app6.docx" xlink:title="DOCX File, 3136 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Calibration plots for logistic regression, random forest, and gradient boosting models across 64 cohorts.</p><media xlink:href="formative_v10i1e88406_app7.docx" xlink:title="DOCX File, 19931 KB"/></supplementary-material></app-group></back></article>