<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e75793</article-id><article-id pub-id-type="doi">10.2196/75793</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Recruitment Source and Participant Retention in a Digital HIV Prevention Intervention for Young Black and Latino Men and Transgender Women: Secondary Analysis of the HealthMpowerment 2.0 Randomized Controlled Trial</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Lin</surname><given-names>Willey Y</given-names></name><degrees>MBIOT</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Choi</surname><given-names>Seul Ki</given-names></name><degrees>MPH, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hirshfield</surname><given-names>Sabina</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mulawa</surname><given-names>Marta I</given-names></name><degrees>PhD, MHS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Watson</surname><given-names>Dovie L</given-names></name><degrees>MD, MSCE</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hightow-Weidman</surname><given-names>Lisa B</given-names></name><degrees>MPH, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Muessig</surname><given-names>Kathryn E</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Bauermeister</surname><given-names>Jos&#x00E9; A</given-names></name><degrees>MPH, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Family and Community Health, School of Nursing, University of Pennsylvania</institution><addr-line>418 Curie Blvd, Rm 237L</addr-line><addr-line>Philadelphia</addr-line><addr-line>PA</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Medicine, SUNY Downstate Health Sciences University</institution><addr-line>Brooklyn</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff3"><institution>School of Nursing, Duke University</institution><addr-line>Durham</addr-line><addr-line>NC</addr-line><country>United States</country></aff><aff id="aff4"><institution>Department of Medicine, Perelman School of Medicine, University of Pennsylvania</institution><addr-line>Philadelphia</addr-line><addr-line>PA</addr-line><country>United States</country></aff><aff id="aff5"><institution>Institute on Digital Health and Innovation, College of Nursing, Florida State University</institution><addr-line>Tallahassee</addr-line><addr-line>FL</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Raymond Guo</surname><given-names>L</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Threats</surname><given-names>Megan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Willey Y Lin, MBIOT, Department of Family and Community Health, School of Nursing, University of Pennsylvania, 418 Curie Blvd, Rm 237L, Philadelphia, PA, 19104, United States, 1 6693339925; <email>willey@nursing.upenn.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>26</day><month>8</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e75793</elocation-id><history><date date-type="received"><day>10</day><month>04</month><year>2025</year></date><date date-type="rev-recd"><day>02</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>03</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Willey Y Lin, Seul Ki Choi, Sabina Hirshfield, Marta I Mulawa, Dovie L Watson, Lisa B Hightow-Weidman, Kathryn E Muessig, Jos&#x00E9; A Bauermeister. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 26.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e75793"/><abstract><sec><title>Background</title><p>Young Black and Latino men who have sex with men and transgender women who have sex with men (YBLMT) experience disproportionate HIV-related health disparities in the United States. Digital health interventions offer scalable HIV prevention and support services for these populations. However, recruitment strategies may influence both sample demographics and participant retention, which is critical for intervention effectiveness.</p></sec><sec><title>Objective</title><p>This study examined whether recruitment source was associated with retention in a mobile health randomized controlled trial and assessed demographic differences across recruitment platforms.</p></sec><sec sec-type="methods"><title>Methods</title><p>Data were drawn from the HealthMpowerment (HMP) 2.0 randomized controlled trial (N=750; July 2020 to September 2022). Participants aged 15 to 29 years were recruited through 4 sources: social media advertisements (n=202), partner-seeking apps (n=394), prior study contacts (n=118), and other sources (eg, community outreach, peer referrals; n=36). Retention was defined as completing at least 3 of 4 follow-up surveys over 12 months, equivalent to completing at least 80% (n=4) of all 5 study measurement occasions, consistent with the United States Preventive Services Task Force (USPSTF) criteria for cohort study follow-up. Chi-square and Fisher exact tests assessed unadjusted retention differences across recruitment sources. Multivariable logistic regression adjusting for age, HIV status, race, ethnicity, and gender identity estimated adjusted odds ratios (aOR) for the association between recruitment source and retention. A sensitivity analysis requiring completion of all 4 follow-up surveys was also conducted. A predictive logistic regression model incorporating Synthetic Minority Oversampling Technique (SMOTE) to address class imbalance further explored demographic and recruitment predictors of retention.</p></sec><sec sec-type="results"><title>Results</title><p>Recruitment source was significantly associated with retention (<italic>&#x03C7;</italic>&#x00B2;<sub>3</sub>=36.62; <italic>P</italic>&#x003C;.001). Participants recruited via social media (184/202, 91.1%) and prior study contacts (103/118, 87.3%) had higher retention than those recruited via partner-seeking apps (288/394, 73.1%) and other sources (23/36, 63.9%). After adjusting for demographics, social media (aOR 2.80, 95% CI 1.51&#x2010;5.18) and study contacts (aOR 2.11, 95% CI 1.14&#x2010;3.92) remained significantly associated with higher retention compared to partner-seeking apps. HIV-positive status (aOR 0.64) and gender-diverse identity (aOR 0.38) were independently associated with lower retention. The sensitivity analysis yielded directionally consistent findings, with the social media advantage remaining significant. The SMOTE-adjusted model improved recall for low-retention participants from 6% (2/33) to 64% (21/33) but reduced overall accuracy from 77.3% (116/150) to 64.7% (97/150).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Recruitment source was associated with both participant characteristics and long-term retention in this digital HIV prevention intervention. Social media and prior research networks yielded higher retention, whereas partner-seeking apps reached populations at higher HIV risk but were associated with lower retention. Targeted retention strategies that account for both recruitment platform and participant characteristics (including HIV status and gender identity) may help optimize engagement and sample diversity in future digital health interventions targeting YBLMT.</p></sec><sec><title>Trial Registration</title><p>ClinicalTrials.gov NCT03678181; https://clinicaltrials.gov/study/NCT03678181</p></sec><sec sec-type="registered-report"><title>International Registered Report Identifier (IRRID)</title><p>RR2-10.2196/24043</p></sec></abstract><kwd-group><kwd>HIV</kwd><kwd>men who have sex with men</kwd><kwd>mHealth</kwd><kwd>Hispanic Americans</kwd><kwd>African Americans</kwd><kwd>recruitment</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Young Black and Latino men who have sex with men and transgender women who have sex with men (YBLMT) in the United States face disproportionate health disparities, including elevated rates of HIV infection and barriers to care [<xref ref-type="bibr" rid="ref1">1</xref>]. These disparities are driven by intersectional stigmas related to race, ethnicity, sexuality, and HIV status, further compounded by geographic and logistical barriers such as living in rural or homophobic communities or being located far from research institutions [<xref ref-type="bibr" rid="ref2">2</xref>]. As our day-to-day lives become increasingly digital, online health interventions offer opportunities to address these challenges by providing better privacy, confidentiality, and accessibility, thereby overcoming many geographical and logistical limitations [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>While online health interventions enhance accessibility, recruitment strategies must also evolve to effectively reach underrepresented populations. Traditional methods, such as venue-based sampling and community outreach, often struggle to engage YBLMT due to concerns about selection bias, confidentiality, stigma, and logistical constraints [<xref ref-type="bibr" rid="ref5">5</xref>]. For example, venue-based recruitment at LGBTQ+ events may exclude individuals who do not openly identify or participate in these spaces.</p><p>Online recruitment methods, including advertising on social media and dating apps, address some of these limitations by offering greater reach and privacy [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. However, they also introduce new challenges, such as the risk of fraudulent responses, competition for ad visibility, and increasing recruitment costs [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Further, targeted digital advertising can drive up costs due to competition with commercial advertisers, particularly when targeting niche populations [<xref ref-type="bibr" rid="ref6">6</xref>]. Balancing cost-effectiveness with strategies that ensure data integrity and minimize selection biases is essential for optimizing online recruitment efforts.</p><p>Beyond cost and accessibility, another critical consideration is how different recruitment sources influence the demographic composition of study samples and participant retention over time. Prior research suggests that recruitment sources not only shape who enrolls in a study but also impact how likely participants are to complete follow-up assessments [<xref ref-type="bibr" rid="ref8">8</xref>]. For example, social media ads tend to attract younger, more tech-savvy participants who may exhibit higher retention [<xref ref-type="bibr" rid="ref8">8</xref>], while dating apps may reach older and more diverse populations, including individuals at higher risk for HIV [<xref ref-type="bibr" rid="ref8">8</xref>]. These differences influence sample diversity and retention, both of which are crucial for ensuring generalizable findings. While previous studies have explored how recruitment sources affect study engagement, limited research has systematically examined their impact on participant retention, particularly in digital health interventions targeting YBLMT.</p><p>Given the potential impact of recruitment strategies on participant diversity and retention, a more systematic understanding of these patterns is necessary for designing effective recruitment strategies. Identifying predictors of retention, such as demographic or recruitment factors linked to sustained participation, can help refine recruitment strategies to improve study retention. While recruitment source itself is not a causal determinant of retention, different platforms may attract participants with varying likelihoods of completing study follow-ups [<xref ref-type="bibr" rid="ref8">8</xref>]. Understanding these recruitment-driven variations can help researchers optimize outreach to improve retention or adjust analyses to account for selection bias when comparing recruitment effectiveness.</p><p>Longitudinal data from the HealthMpowerment (HMP) 2.0 study provided an opportunity to explore predictors of study retention. HMP 2.0 is a mobile health intervention evaluated through a randomized controlled trial (RCT) designed to reduce intersectional stigma and improve HIV-related outcomes among YBLMT. The study used multiple recruitment sources, including paid ads on social media and partner-seeking apps, email listservs of past study contacts, and traditional community-based methods.</p><p>The objective of this study was to examine whether recruitment source was associated with participant retention in the HMP 2.0 RCT. We additionally explored demographic differences across recruitment sources and assessed demographic and recruitment predictors of retention. Ultimately, these insights aim to enhance inclusivity, optimize retention in online health interventions, and ensure that recruitment strategies are cost-effective and tailored to the needs of diverse populations.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Procedures</title><p>All data were collected as part of the 3-arm RCT evaluating HMP 2.0. The 3 arms were the tailored information-only control (attention control), researcher-created HMP network (intervention arm 1), and peer-referral network (intervention arm 2). Details of the original study design and intervention are described in the published protocol [<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Participants were recruited between July 2020 and September 2022 through 4 primary sources: paid ads on social media platforms (eg, Facebook, Instagram, X [formerly Twitter]); paid ads on partner-seeking apps (eg, Jack&#x2019;d, Scruff, Grindr); study contacts (ie, individuals who had previously participated in research conducted by the study team or affiliated projects and had consented to be recontacted about future studies through institutional email listservs); and other sources that included in-person recruitment, peer referrals, community outreach, and postings on LGBTQ+ forums. Ads on social media platforms and partner-seeking apps appeared in users&#x2019; feeds, inboxes, or app interfaces and directed interested individuals to an online Qualtrics screener to verify eligibility.</p><p>To be eligible for the study, individuals had to be between the ages of 15 and 29 years (inclusive) at the time of screening, identify as Black or African American and/or Latino or Hispanic (except participants referred by those in Intervention arm 2), reside in the United States, and speak and read English. They were also required to have regular access to a smartphone and either report anal intercourse in the past 6 months or have been recruited from a partner-seeking app. The study included individuals assigned male sex at birth, with no restrictions on their current gender identity.</p><p>Interested individuals completed an online screening survey [<xref ref-type="bibr" rid="ref10">10</xref>], and those who met all study eligibility criteria were emailed a unique link to the study&#x2019;s informed consent form. Upon providing consent, participants completed a 30 to 50-minute baseline computer-assisted self-interview (CASI) survey.</p><p>To maintain data integrity, a thorough fraud detection process was implemented that involved reviewing IP addresses, geolocation data, and comparing responses between the screening and baseline surveys. Additionally, participant data were cross-checked with the study&#x2019;s participant database to detect duplicate or fraudulent entries. This process effectively identified and removed duplicate or fraudulent entries.</p><p>Participants who passed the validation checks were randomized into 1 of the 3 study arms using a computer-generated blocked randomization procedure stratified by HIV status. They were then provided a link to download the study app from their respective app store (ie, Apple&#x2019;s App Store or Google Play). To support retention across the 12-month study period, participants received regular reminders via email, text messages, and in-app notifications. These efforts aimed to maximize completion rates for follow-up surveys conducted at 3, 6, 9, and 12 months.</p><p>Participants could receive up to US $490 in Amazon gift card incentives for completing study surveys and other study-related activities. The breakdown of incentives includes up to US $280 for completing the study surveys, up to US $140 for completing study test kits, up to US $20 for successfully referring participants to the study (intervention arm 2 only), and US $50 for completing an in-depth interview, if selected.</p></sec><sec id="s2-2"><title>Measures</title><sec id="s2-2-1"><title>Recruitment Sources</title><p>The primary independent variable was recruitment source, categorized as social media, partner-seeking apps, study contacts, and other sources. Each participant&#x2019;s recruitment source was recorded at intake. Participants recruited via social media platforms (eg, Facebook, Instagram) and partner-seeking apps (eg, Jack&#x2019;d, Scruff, Grindr) were classified according to the platform on which the study ad appeared. Platforms with fewer than 10 participants were grouped into their respective broad category rather than analyzed as stand-alone groups, as small cell sizes would have precluded meaningful statistical comparisons. Other sources comprised community-based recruitment methods including in-person outreach, peer referrals, and postings on LGBTQ+ forums.</p></sec><sec id="s2-2-2"><title>Retention Metrics</title><p>Participant retention was defined as the completion of follow-up surveys at 3, 6, 9, and 12 months. For this analysis, retention was dichotomized into 2 groups: high retention (completion of at least 3 of 4 follow-up surveys) and low retention (completion of fewer than 3 surveys). Because all 750 enrolled participants completed the baseline assessment, this threshold equates to completing at least 4 (80%) of 5 total study measurement occasions, consistent with the United States Preventive Services Task Force (USPSTF) Quality Rating Criteria defining &#x201C;Good&#x201D; quality follow-up for cohort studies as greater than 80% retention [<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>Because there was an imbalance in retention categories, with high-retention participants significantly outnumbering low-retention participants, the Synthetic Minority Oversampling Technique (SMOTE) [<xref ref-type="bibr" rid="ref12">12</xref>] was applied to adjust for class imbalance during model development. SMOTE was applied only to the training dataset, while the testing dataset retained its original class distribution to ensure unbiased evaluation of model performance.</p></sec><sec id="s2-2-3"><title>Demographic Characteristics</title><p>Demographic characteristics were assessed using structured self-report items in the screening and baseline surveys. To ensure statistical robustness and consistency in analysis, these variables were categorized accordingly. Age was grouped into 3 categories: 15 to 19, 20 to 24, and 25 to 29 years. Ethnicity was classified as Hispanic or non-Hispanic. For race, because the survey allowed participants to select multiple racial identities, a binary categorization was used: participants who selected only &#x201C;White&#x201D; were categorized as &#x201C;White,&#x201D; while all others, including those who selected multiple racial categories, were grouped as &#x201C;Non-White or Multiracial.&#x201D; This approach was chosen to avoid misclassifying multiracial participants and is consistent with analytic approaches used in studies with similar survey instruments. Hispanic ethnicity was retained as a separate variable, as Latino identity is an ethnic rather than a racial category and captures a distinct dimension of participant background. Gender identity was categorized such that individuals identifying exclusively as &#x201C;man&#x201D; were classified as &#x201C;cis,&#x201D; while all others, including transgender women, nonbinary individuals, and other gender identities, were grouped as &#x201C;gender diverse.&#x201D; Individual gender identity subgroups were too small to support stable separate estimates. Finally, HIV status was classified as either HIV-positive or HIV-negative.</p><p>These standardized categorizations ensured comparability across analyses, improving descriptive statistics and logistic regression modeling while addressing dataset imbalances.</p></sec></sec><sec id="s2-3"><title>Data Analysis</title><sec id="s2-3-1"><title>Overview</title><p>All analyses were conducted with Python 3 in Google Colaboratory, using pandas [<xref ref-type="bibr" rid="ref13">13</xref>], scikit-learn [<xref ref-type="bibr" rid="ref14">14</xref>], imbalanced-learn [<xref ref-type="bibr" rid="ref15">15</xref>], and statsmodels [<xref ref-type="bibr" rid="ref16">16</xref>]. These libraries were selected to support data preprocessing, statistical analysis, and machine learning (ML)&#x2013;based modeling. The dataset was first summarized using descriptive statistics to provide an overview of participant characteristics and retention patterns.</p><p>The analysis was structured into 3 components: primary analysis, secondary analysis, and exploratory analysis, each designed to address its specific objectives.</p></sec><sec id="s2-3-2"><title>Primary Analysis</title><p>The primary analysis examined the association between recruitment source and participant retention. We used Pearson&#x2019;s chi-squared test and Fisher exact test to assess retention differences across the 4 recruitment sources: social media, partner-seeking apps, study contacts, and other sources. These statistical tests were chosen for their suitability with categorical data and robustness in handling imbalanced sample sizes. Then, odds ratios (OR) and corresponding 95% CIs were calculated. This analysis aimed to determine whether specific recruitment sources were associated with higher participant retention rates.</p><p>To evaluate whether the association between recruitment source and retention was independent of demographic differences across sources, a multivariable logistic regression was conducted using statsmodels. The model specified retention (&#x2265;3 of 4 follow-up surveys) as the outcome and included recruitment source as the primary exposure variable, with age, HIV status, race, ethnicity, and gender identity included as covariates. Partner-seeking apps served as the reference category for recruitment source, given their representational dominance in the sample (n=394, 52.5%). Reference categories for covariates were as follows: 15 to 19 years (age), HIV-negative (HIV status), non-White or multiracial (race), non-Hispanic (ethnicity), and cisgender man (gender identity). The results are reported as adjusted odds ratios (aOR) with 95% CIs and <italic>P</italic> values. Adjusted pairwise ORs between all recruitment source pairs were obtained by respecifying the model with each category as the reference in turn.</p></sec><sec id="s2-3-3"><title>Sensitivity Analysis</title><p>To assess the robustness of the primary findings to the choice of retention threshold, a sensitivity analysis was conducted using a stricter definition, classifying high retention as the completion of all 4 follow-up surveys. The chi-square test and multivariable logistic regression were repeated using identical model specifications and reference categories as the primary analysis.</p></sec><sec id="s2-3-4"><title>Secondary Analysis</title><p>The secondary analysis explored demographic patterns across the largest recruitment sources, specifically comparing social media and partner-seeking apps. These 2 sources were prioritized because they accounted for the majority of participants, ensuring sufficient statistical power. Recruitment categories with lower representation (ie, study contacts and other sources) were excluded as they would have limited statistical power for meaningful comparisons. Detailed recruitment source distributions are provided in the <italic>Results</italic> section. We used Pearson chi-squared test and Fisher exact test to examine differences in age, ethnicity, race, HIV status, and gender identity distributions between participants recruited through social media compared to partner-seeking apps.</p></sec><sec id="s2-3-5"><title>Exploratory Analysis</title><p>Advances in AI and ML have revolutionized data analysis across disciplines, yet their application in social and behavioral sciences remains relatively underexplored [<xref ref-type="bibr" rid="ref17">17</xref>]. Traditional inferential models, which emphasize statistical significance and hypothesis testing, dominate behavioral health research [<xref ref-type="bibr" rid="ref18">18</xref>]. In contrast, predictive models, such as ML-driven logistic regression, focus on identifying patterns in complex datasets and evaluating how well models generalize to new data. By leveraging these strengths, predictive modeling offers an opportunity to enhance the identification of factors influencing participant retention.</p><p>To explore predictors of study retention, we developed a predictive logistic regression model incorporating recruitment source, age, ethnicity, race, HIV status, and gender identity as predictors. In this exploratory analysis, the model was used primarily to identify patterns associated with retention rather than to test formal causal hypotheses. We used logistic regression coefficients (&#x03B2;) to assess both the direction and relative magnitude of associations between predictors and retention. Because these coefficients provide interpretable, directional estimates (indicating whether a factor increases or decreases the likelihood of retention), this approach allows for a direct understanding of how specific characteristics influence longitudinal participation.</p><p>To develop and evaluate the model, we split the dataset into a training set (n=600, 80%) and a testing set (n=150, 20%) [<xref ref-type="bibr" rid="ref19">19</xref>]. The training set was used to train the predictive model, while the testing set provided an independent evaluation of model performance on new, unseen data [<xref ref-type="bibr" rid="ref18">18</xref>]. High-retention participants comprised approximately 80% (598/750) of the sample, creating a class imbalance that could bias models toward predicting retention. To address this, we tested two logistic regression models:</p><list list-type="order"><list-item><p>Baseline model: Trained on the original, imbalanced dataset.</p></list-item><list-item><p>SMOTE-adjusted model: Trained on a dataset where the minority class (low retention) was oversampled using the SMOTE, which generates synthetic samples for the minority class by interpolating between existing data points rather than simply duplicating cases.</p></list-item></list><p>Comparing the 2 models allowed us to assess whether balancing the training data improved sensitivity to low-retention participants, and at what cost to overall accuracy, since oversampling may introduce artificial patterns that do not generalize well [<xref ref-type="bibr" rid="ref20">20</xref>]. The original class distribution was preserved in the testing set throughout.</p><p>To systematically compare model performance, we used four standard ML evaluation metrics:</p><list list-type="bullet"><list-item><p>Precision: The proportion of correctly predicted high or low retention cases out of all cases predicted as high or low retention.</p></list-item><list-item><p>Recall<italic>:</italic> The proportion of actual high or low retention cases identified by the model.</p></list-item><list-item><p><italic>F</italic><sub>1</sub>-score: A harmonic mean of precision and recall, balancing sensitivity and specificity.</p></list-item><list-item><p>Accuracy: The overall proportion of correctly classified cases across all retention levels.</p></list-item></list><p>By splitting the data into training and testing sets, this analysis ensured that model performance reflected its predictive power rather than overfitting to the dataset [<xref ref-type="bibr" rid="ref19">19</xref>]. This predictive modeling framework prioritizes generalizability, allowing for more reliable identification of potential retention predictors in future research. Additionally, SMOTE corrects for class imbalances, reducing bias that might otherwise affect standard logistic regression models.</p><p>This hypothesis-generating framework complements traditional inferential methods by identifying emerging patterns and potential predictors of retention that may warrant further investigation in future studies.</p></sec></sec><sec id="s2-4"><title>Ethical Considerations</title><p>The HMP 2.0 RCT was approved by the University of Pennsylvania institutional review board (IRB; protocol #829805) and registered on ClinicalTrials.gov (NCT03678181). All participants provided informed consent prior to their participation, with individuals aged 15 to 17 years granted a waiver of parental consent to protect their privacy as sexual minority youth. To ensure confidentiality, all collected data were deidentified prior to analysis.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Participant Characteristics</title><p>A total of 750 participants were enrolled in the study between July 2020 and September 2022. The majority (n=445, 59.3%) were aged 25 to 29 years, followed by 236 (31.5%) participants aged 20 to 24 years and 69 (9.2%) participants aged 15 to 19 years. In terms of ethnicity, 318 (42.4%) participants identified as Hispanic, while 432 (57.6%) participants identified as non-Hispanic. Regarding race, 609 (81.2%) participants identified as non-White or multiracial.</p><p>Among study participants, 230 (30.7%) reported living with HIV, while 520 (69.3%) were HIV-negative. Most participants (n=666, 88.8%) identified as cisgender men, while 84 (11.2%) participants were categorized as gender diverse, including transgender women (n=16), nonbinary individuals (n=64), and other gender identities (n=4).</p><p>Participants were recruited through 4 primary sources. The largest proportion (n=394, 52.5%) were enrolled via partner-seeking apps. Social media platforms accounted for 202 (26.9%) participants, followed by prior study contacts, which contributed 118 participants (15.7%). The remaining 36 participants (4.8%) were recruited through other sources, including peer referrals, in-person outreach, and LGBTQ+ forums. Participant demographics and recruitment sources are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Demographic characteristics and recruitment sources of young Black and Latino men who have sex with men and transgender women who have sex with men (YBLMT) who participated in HealthMpowerment (HMP) 2.0 (N=750).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Participants, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Age (y)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>15&#x2010;19</td><td align="left" valign="top">69 (9.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>20&#x2010;24</td><td align="left" valign="top">236 (31.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>25&#x2010;29</td><td align="left" valign="top">445 (59.3)</td></tr><tr><td align="left" valign="top" colspan="2">Ethnicity</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hispanic</td><td align="left" valign="top">318 (42.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Non-Hispanic</td><td align="left" valign="top">432 (57.6)</td></tr><tr><td align="left" valign="top" colspan="2">Race</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>White</td><td align="left" valign="top">141 (18.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Non-White or multiracial</td><td align="left" valign="top">609 (81.2)</td></tr><tr><td align="left" valign="top" colspan="2">HIV status</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>HIV-positive</td><td align="left" valign="top">230 (30.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>HIV-negative</td><td align="left" valign="top">520 (69.3)</td></tr><tr><td align="left" valign="top" colspan="2">Gender identity</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cisgender man</td><td align="left" valign="top">666 (88.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gender diverse</td><td align="left" valign="top">84 (11.2)</td></tr><tr><td align="left" valign="top" colspan="2">Recruitment source</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Social media</td><td align="left" valign="top">202 (26.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Partner-seeking apps</td><td align="left" valign="top">394 (52.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Study contacts</td><td align="left" valign="top">118 (15.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other sources</td><td align="left" valign="top">36 (4.8)</td></tr></tbody></table></table-wrap><p>Survey completion varied across participants. The majority (n=514, 68.5%) completed all 4 follow-up surveys. An additional 84 (11.2%) participants completed 3 surveys, meeting the high retention threshold. The remaining 152 (20.3%) participants were classified as low retention, including 36 (4.8%) who completed 2 surveys, 36 (4.8%) who completed 1 survey, and 80 (10.7%) who did not complete any follow-up surveys.</p></sec><sec id="s3-2"><title>Primary Analysis: Association Between Recruitment Source and Participant Retention</title><p>A significant association was observed between recruitment source and participant retention (<italic>&#x03C7;</italic>&#x00B2;<sub>3</sub>=36.62, N=750; <italic>P</italic>&#x003C;.001). Retention rates varied across recruitment sources. Retention rates were the highest among participants recruited via social media and study contacts, while those recruited from partner-seeking apps and other sources had lower retention rates.</p><p>Among participants recruited via social media, 184 (91.1%) participants met the high retention criterion (completing at least 3 follow-up surveys; <xref ref-type="fig" rid="figure1">Figure 1</xref>). Similarly, 103 (87.3%) participants recruited via study contacts were highly retained. In contrast, participants recruited via partner-seeking apps had a retention rate of 288 (73.1%) participants, while those recruited via other sources had a retention rate of 23 (63.9%) participants.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flow of total study sample stratified by the 4 primary recruitment sources. Participant progression is tracked sequentially across 4 follow-up intervals, detailing the number of participants retained and those who dropped out at each specific stage. The highlighted tier (in green) represents the study&#x2019;s predefined threshold for high retention.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e75793_fig01.png"/></fig><p>Fisher exact tests revealed significant differences in retention between social media and other recruitment sources (<xref ref-type="table" rid="table2">Table 2</xref>). Retention was significantly higher among participants recruited via social media compared to those recruited via partner-seeking apps (OR 3.76, 95% CI 2.21&#x2010;6.41; <italic>P</italic>&#x003C;.001) and other sources (OR 5.78, 95% CI 2.51&#x2010;13.31; <italic>P</italic>&#x003C;.001). Retention did not significantly differ between social media and study contacts (OR 1.49, 95% CI 0.72&#x2010;3.08; <italic>P</italic>=.34). Participants recruited via study contacts also had significantly higher retention than those recruited via partner-seeking apps (OR 2.53, 95% CI 1.41&#x2010;4.54; <italic>P</italic>=.001) and other sources (OR 3.88, 95% CI 1.63&#x2010;9.26; <italic>P</italic>=.003). Retention did not significantly differ between partner-seeking apps and other sources (OR 1.54, 95% CI 0.75&#x2010;3.14; <italic>P</italic>=.25).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Retention rates and pairwise comparisons by recruitment source.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison</td><td align="left" valign="bottom">OR<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Social media vs partner-seeking apps</td><td align="left" valign="top">3.76 (2.21-6.41)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Social media vs other sources</td><td align="left" valign="top">5.78 (2.51-13.31)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Social media vs study contacts</td><td align="left" valign="top">1.49 (0.72-3.08)</td><td align="left" valign="top">.34</td></tr><tr><td align="left" valign="top">Study contacts vs partner-seeking apps</td><td align="left" valign="top">2.53 (1.41-4.54)</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top">Study contacts vs other sources</td><td align="left" valign="top">3.88 (1.63-9.26)</td><td align="left" valign="top">.003</td></tr><tr><td align="left" valign="top">Partner-seeking apps vs other sources</td><td align="left" valign="top">1.54 (0.75-3.14)</td><td align="left" valign="top">.25</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>OR: odds ratio.</p></fn></table-wrap-foot></table-wrap><p>To evaluate whether these differences were independent of demographic variation across recruitment sources, a multivariable logistic regression was conducted adjusting for age, HIV status, race, ethnicity, and gender identity (N=750; pseudo <italic>R</italic>&#x00B2;=0.092; Akaike information criterion [AIC]=706.6). After adjustment, participants recruited via social media (aOR 2.80, 95% CI 1.51&#x2010;5.18; <italic>P</italic>=.001) and study contacts (aOR 2.11, 95% CI 1.14&#x2010;3.92; <italic>P</italic>=.02) remained significantly associated with higher odds of retention compared to those recruited via partner-seeking apps. Other sources did not differ significantly from partner-seeking apps after adjustment (aOR 0.61, 95% CI 0.28&#x2010;1.31; <italic>P</italic>=.21). Full model results are presented in <xref ref-type="table" rid="table3">Table 3</xref>.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Multivariable logistic regression: adjusted association between recruitment source and retention (&#x2265;3 of 4 surveys; N=750; model fit: pseudo <italic>R</italic>&#x00B2;=0.092; AIC=706.6).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Predictor</td><td align="left" valign="bottom">aOR<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Recruitment source (reference: partner-seeking apps)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Social media</td><td align="left" valign="top">2.80 (1.51-5.18)</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Study contacts</td><td align="left" valign="top">2.11 (1.14-3.92)</td><td align="left" valign="top">.018</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other sources</td><td align="left" valign="top">0.61 (0.28-1.31)</td><td align="left" valign="top">.21</td></tr><tr><td align="left" valign="top" colspan="3">Age (y; reference: 15&#x2010;19)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>20&#x2010;24</td><td align="left" valign="top">0.48 (0.21-1.08)</td><td align="left" valign="top">.08</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>25&#x2010;29</td><td align="left" valign="top">0.95 (0.42-2.13)</td><td align="left" valign="top">.90</td></tr><tr><td align="left" valign="top" colspan="3">HIV status (reference: HIV-negative)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>HIV-positive</td><td align="left" valign="top">0.64 (0.42-0.96)</td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top" colspan="3">Race (reference: Non-White or multiracial)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>White</td><td align="left" valign="top">1.22 (0.60-2.45)</td><td align="left" valign="top">.59</td></tr><tr><td align="left" valign="top" colspan="3">Ethnicity (reference: Non-Hispanic)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hispanic</td><td align="left" valign="top">1.20 (0.74-1.95)</td><td align="left" valign="top">.45</td></tr><tr><td align="left" valign="top" colspan="3">Gender identity (reference: cisgender man)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gender diverse</td><td align="left" valign="top">0.38 (0.23-0.63)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>aOR: adjusted odds ratio.</p></fn></table-wrap-foot></table-wrap><p>Among covariates, HIV-positive status was independently associated with lower odds of retention (aOR 0.64, 95% CI 0.42&#x2010;0.96; <italic>P</italic>=.03). Gender-diverse participants also had significantly lower odds of retention compared to cisgender men (aOR 0.38, 95% CI 0.23&#x2010;0.63; <italic>P</italic>&#x003C;.001). Age, race, and ethnicity were not independently associated with retention after adjustment.</p><p>Adjusted pairwise comparisons across all 4 recruitment source pairs are presented in <xref ref-type="table" rid="table4">Table 4</xref>. Other sources had significantly lower adjusted odds of retention compared to both social media (aOR 0.22, 95% CI 0.09&#x2010;0.54; <italic>P</italic>=.001) and study contacts (aOR 0.29, 95% CI 0.12&#x2010;0.72; <italic>P</italic>=.007). Social media and study contacts did not differ significantly from each other after adjustment (aOR 0.76, 95% CI 0.35&#x2010;1.61; <italic>P</italic>=.47).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Adjusted pairwise odds ratios for retention (&#x2265;3 of 4 surveys) across all recruitment source pairs, adjusted for age, HIV status, race, ethnicity, and gender identity.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison</td><td align="left" valign="bottom">aOR<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Social media vs partner-seeking apps</td><td align="left" valign="top">2.80 (1.51-5.18)</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top">Study contacts vs partner-seeking apps</td><td align="left" valign="top">2.11 (1.14-3.92)</td><td align="left" valign="top">.02</td></tr><tr><td align="left" valign="top">Other sources vs partner-seeking apps</td><td align="left" valign="top">0.61 (0.28-1.31)</td><td align="left" valign="top">.21</td></tr><tr><td align="left" valign="top">Study contacts vs social media</td><td align="left" valign="top">0.76 (0.35-1.61)</td><td align="left" valign="top">.47</td></tr><tr><td align="left" valign="top">Other sources vs social media</td><td align="left" valign="top">0.22 (0.09-0.54)</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top">Other sources vs study contacts</td><td align="left" valign="top">0.29 (0.12-0.72)</td><td align="left" valign="top">.007</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>aOR: adjusted odds ratio.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Sensitivity Analysis</title><p>Using a strict definition (completion of all 4 surveys; n=514, 68.5% of the participants), retention rates were 83.2% (168/202) for social media, 72.9% (86/118) for study contacts, 61.2% (241/394) for partner-seeking apps, and 52.8% (19/36) for other sources. The overall <italic>&#x03C7;</italic><sup>2</sup> remained significant (<italic>&#x03C7;</italic>&#x00B2;<sub>3</sub>=35.15, N=750; <italic>P</italic>&#x003C;.001). Social media remained significantly associated with higher odds of retention compared to partner-seeking apps (aOR 2.79, 95% CI 1.70&#x2010;4.57; <italic>P</italic>&#x003C;.001). The advantage for study contacts was attenuated and no longer reached significance (aOR 1.58, 95% CI 0.97&#x2010;2.55; <italic>P</italic>=.07).</p></sec><sec id="s3-4"><title>Secondary Analysis: Demographic Patterns Across the Two Largest Recruitment Sources</title><sec id="s3-4-1"><title>Overview</title><p>Demographic characteristics varied significantly between participants recruited via social media and those recruited via partner-seeking apps. Analyses were limited to these 2 sources (n=596), as they accounted for the majority of participants. Demographic patterns across the 2 largest recruitment sources are summarized in <xref ref-type="table" rid="table5">Table 5</xref>.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Demographic differences between participants recruited via social media and partner-seeking apps (n=596).</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Social media (n=202), n (%)</td><td align="left" valign="bottom">Partner-seeking apps (n=394), n (%)</td><td align="left" valign="bottom">Chi-square (<italic>df</italic>)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">33.88 (2)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>15&#x2010;19</td><td align="left" valign="top">29 (14.4)</td><td align="left" valign="top">11 (2.8)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>20&#x2010;24</td><td align="left" valign="top">73 (36.1)</td><td align="left" valign="top">122 (31.0)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>25&#x2010;29</td><td align="left" valign="top">100 (49.5)</td><td align="left" valign="top">261 (66.2)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Race</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">116.27 (1)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>White</td><td align="left" valign="top">82 (40.6)</td><td align="left" valign="top">20 (5.1)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Non-White or multiracial</td><td align="left" valign="top">120 (59.4)</td><td align="left" valign="top">374 (94.9)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Ethnicity</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">152.56 (1)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hispanic</td><td align="left" valign="top">153 (75.7)</td><td align="left" valign="top">90 (22.8)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Non-Hispanic</td><td align="left" valign="top">49 (24.3)</td><td align="left" valign="top">304 (77.2)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top">HIV status</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">88.91 (1)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>HIV-positive</td><td align="left" valign="top">11 (5.4)</td><td align="left" valign="top">171 (43.4)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>HIV-negative</td><td align="left" valign="top">191 (94.6)</td><td align="left" valign="top">223 (56.6)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Gender identity</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">.77 (1)</td><td align="left" valign="top">.38</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cisgender man</td><td align="left" valign="top">183 (90.6)</td><td align="left" valign="top">346 (87.8)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gender diverse</td><td align="left" valign="top">19 (9.4)</td><td align="left" valign="top">48 (12.2)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr></tbody></table></table-wrap></sec><sec id="s3-4-2"><title>Age</title><p>Participants recruited via social media (n=202) were younger on average, including those aged 15 to 19 years (n=29, 14.4%), 20 to 24 years (n=73, 36.1%), and 25 to 29 years (n=100, 49.5%). By contrast, participants recruited through partner-seeking apps (n=394) were generally older, including those aged 15 to 19 years (n=11, 2.8%), 20 to 24 years (n=122, 31.0%), and 25 to 29 years (n=261, 66.2%). A chi-square test confirmed significant differences in age distribution between the 2 groups (<italic>&#x03C7;</italic>&#x00B2;<sub>2</sub>=33.88, n=596; <italic>P</italic>&#x003C;.001).</p></sec><sec id="s3-4-3"><title>Race or Ethnicity</title><p>A higher proportion of participants recruited via social media identified as White (82/202, 40.6%), whereas a higher proportion of those recruited via partner-seeking apps identified as non-White or multiracial (374/394, 94.9%). Similarly, social media recruitment yielded a predominantly Hispanic participant pool (153/202, 75.7%), whereas participants recruited from partner-seeking apps were less likely to identify as Hispanic (90/394, 22.8%). Chi-squared tests confirmed significant differences in both race (<italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=116.27, n=596; <italic>P</italic>&#x003C;.001) and ethnicity (<italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=152.56, n=596; <italic>P</italic>&#x003C;.001) distributions between the 2 recruitment sources.</p></sec><sec id="s3-4-4"><title>HIV Status</title><p>HIV-negative participants accounted for the vast majority (191/202, 94.6%) of those recruited via social media, whereas a higher proportion (171/394, 43.4%) of the participants recruited via partner-seeking apps were HIV-positive. Chi-square test results confirmed that HIV status distributions differed significantly between recruitment sources (<italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=88.91, n=596; <italic>P</italic>&#x003C;.001).</p></sec><sec id="s3-4-5"><title>Gender Identity</title><p>The distribution of gender identity was largely similar across both recruitment sources, with cisgender men comprising 90.6% (183/202) of participants recruited through social media and 87.8% (346/394) of those recruited through partner-seeking apps. The chi-square test found no significant difference in gender distribution between the 2 recruitment sources (<italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=0.77, n=596; <italic>P</italic>=.38).</p></sec></sec><sec id="s3-5"><title>Exploratory Analysis: Predictors of Participant Retention</title><p>The baseline model achieved an accuracy of 77.3%. It correctly classified 97.4% (114/117) of the participants with high retention but only 6.1% (2/33) of the participants with low retention. The resulting <italic>F</italic><sub>1</sub>-score for low-retention participants was 0.11, reflecting limited precision and recall for this group.</p><p>To improve sensitivity in classifying participants with low retention, the SMOTE-adjusted model was trained on a dataset with an artificially balanced distribution of retention groups. This model achieved a lower overall accuracy of 64.7%. However, recall for low-retention participants increased from 6.1% (2/33) to 63.6% (21/33), and the <italic>F</italic><sub>1</sub>-score for this group improved from 0.11 to 0.44. Recall for high-retention participants decreased from 97.4% (114/117) to 65.0% (76/117). Full performance metrics for both models are presented in <xref ref-type="table" rid="table6">Table 6</xref>.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Comparison of model performance.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Accuracy, n/N (%)</td><td align="left" valign="bottom">Precision (low retention), n/N (%)</td><td align="left" valign="bottom">Recall (low retention), n/N (%)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (low retention)</td><td align="left" valign="bottom">Precision (high retention), n/N (%)</td><td align="left" valign="bottom">Recall (high retention), n/N (%)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (high retention)</td></tr></thead><tbody><tr><td align="left" valign="top">Baseline</td><td align="left" valign="top">116/150 (77.3)</td><td align="left" valign="top">2/5 (40.0)</td><td align="left" valign="top">2/33 (6.1)</td><td align="left" valign="top">0.11</td><td align="left" valign="top">114/145 (78.6)</td><td align="left" valign="top">114/117 (97.4)</td><td align="left" valign="top">0.87</td></tr><tr><td align="left" valign="top">SMOTE<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup>-adjusted</td><td align="left" valign="top">97/150 (64.7)</td><td align="left" valign="top">21/62 (33.9)</td><td align="left" valign="top">21/33 (63.6)</td><td align="left" valign="top">0.44</td><td align="left" valign="top">76/88 (86.4)</td><td align="left" valign="top">76/117 (65.0)</td><td align="left" valign="top">0.74</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>SMOTE: Synthetic Minority Oversampling Technique.</p></fn></table-wrap-foot></table-wrap><p>Beyond overall model performance, logistic regression coefficients were examined to assess the relative contribution of each predictor relative to its reference group (<xref ref-type="table" rid="table7">Table 7</xref>). These coefficients represent the change in the log-odds of high retention when moving from the reference category to the comparison category for each predictor variable, holding all other variables constant. Because all predictors in the model are categorical, their coefficients should be interpreted as the relative likelihood of high retention compared to the reference group. Recruitment source remained the strongest predictor, with social media (&#x03B2;=1.45) and study contacts (&#x03B2;=1.29) showing the largest positive associations with retention compared to partner-seeking apps. These findings were directionally consistent with the adjusted inferential analysis. HIV-positive status (&#x03B2;=&#x2212;.37) and gender-diverse identity (&#x03B2;=&#x2212;0.75) showed negative associations with retention, also consistent with the inferential model. The coefficient for other sources (<italic>&#x03B2;</italic>=0.74) was directionally inconsistent with the adjusted regression, likely reflecting instability in the small subsample available for model training (n&#x2248;29 after the 80/20 split); the adjusted logistic regression is the more reliable estimate for that comparison.</p><table-wrap id="t7" position="float"><label>Table 7.</label><caption><p>Synthetic Minority Oversampling Technique (SMOTE)&#x2013;adjusted logistic regression coefficients for predictors of participant retention (&#x2265;3 of 4 surveys).</p></caption><table id="table7" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison</td><td align="left" valign="bottom">Coefficient</td></tr></thead><tbody><tr><td align="left" valign="top">Social media vs partner-seeking apps</td><td align="left" valign="top">1.45</td></tr><tr><td align="left" valign="top">Study contacts vs partner-seeking apps</td><td align="left" valign="top">1.29</td></tr><tr><td align="left" valign="top">White vs non-White or multiracial</td><td align="left" valign="top">0.82</td></tr><tr><td align="left" valign="top">Other sources vs partner-seeking apps</td><td align="left" valign="top">0.74</td></tr><tr><td align="left" valign="top">Age 25&#x2010;29 vs age 15&#x2010;19</td><td align="left" valign="top">0.54</td></tr><tr><td align="left" valign="top">Hispanic vs non-Hispanic</td><td align="left" valign="top">0.07</td></tr><tr><td align="left" valign="top">Age 20&#x2010;24 vs age 15&#x2010;19</td><td align="left" valign="top">0.00</td></tr><tr><td align="left" valign="top">HIV-positive vs HIV-negative</td><td align="left" valign="top">&#x2013;0.37</td></tr><tr><td align="left" valign="top">Gender diverse vs cisgender man</td><td align="left" valign="top">&#x2013;0.75</td></tr></tbody></table></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>Our findings demonstrate that recruitment source is associated with participant retention in online health interventions for YBLMT. By evaluating how different platforms correspond to retention outcomes, we highlight the need for strategic recruitment planning that goes beyond sample diversity to consider sustained participation over time.</p><p>In line with prior research [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>], our primary analysis found that retention rates differed significantly across recruitment sources. Participants recruited via social media and prior study contacts maintained the highest retention (184/202, 91.1% and, 103/118, 87.3%, respectively). This may reflect platform norms: social media users often engage in longitudinal interactions (such as browsing, commenting, and following content over time), which may translate more effectively to long-term study participation [<xref ref-type="bibr" rid="ref21">21</xref>]. The higher retention observed among prior study contacts likely reflects preexisting institutional trust and an established rapport with the research team [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. Individuals who have previously participated in research and opted into a recontact listserv may possess a higher baseline motivation and a more favorable view of the research process, significantly increasing their likelihood of completing longitudinal follow-up.</p><p>In contrast, the lower retention (288/394, 73.1%) observed among participants recruited via partner-seeking apps may reflect the &#x201C;low-friction&#x201D; and highly goal-oriented nature of those platforms that may not align with sustained engagement [<xref ref-type="bibr" rid="ref25">25</xref>]. Users may engage with a recruitment ad out of momentary curiosity or during a brief window of app usage without the baseline commitment or institutional trust found in social media networks or prior research listservs. This suggests that while apps are excellent for reaching high-risk populations, they may require more intensive &#x201C;onboarding&#x201D; to bridge the gap between initial click and long-term retention. Community-based recruitment (comprising in-person outreach, peer referrals, and LGBTQ+ forum postings) was associated with the lowest adjusted retention across all 4 sources, which may reflect greater structural or logistical barriers to sustained digital study participation among community-recruited participants and warrants further investigation.</p><p>Retention differences may also reflect the distinct participant characteristics associated with each recruitment source, as demonstrated in our secondary analysis. Social media recruitment yielded a younger, predominantly Hispanic, and HIV-negative cohort, whereas partner-seeking apps recruited older, non-White, and HIV-positive participants. Although retention rates were lower among participants recruited via partner-seeking apps, continued recruitment through these platforms is warranted, given that they reach large populations living with or vulnerable to HIV. Allocating additional resources for follow-up efforts, such as tailoring reminder strategies to participants&#x2019; recruitment platforms, may support retention in these groups. By contrast, the high retention observed among prior study contacts suggests that formalized participant referral programs are a useful strategy for sustaining retention in future studies. These patterns highlight the intersectional nature of recruitment dynamics and the importance of adjusting for potential confounders when estimating the independent association between recruitment source and retention.</p><p>Participants recruited via partner-seeking apps represented a cohort that was significantly older, more racially diverse, and more likely to be living with HIV compared to those from social media. These overlapping identities and statuses may correlate with unique structural demands (such as navigating health care systems for HIV care or differing digital engagement habits) that likely influenced their lower retention rates.</p><p>After adjusting for age, HIV status, race, ethnicity, and gender identity, social media and study contacts remained significantly associated with higher retention compared to partner-seeking apps, indicating that the retention advantage observed in these channels is not fully explained by the demographic differences in who they recruit. Beyond recruitment source, 2 demographic factors were independently associated with lower retention after adjustment. HIV-positive participants had significantly lower adjusted odds of retention (aOR 0.64), consistent with evidence that individuals managing HIV face competing health demands, stigma-related burdens, and structural barriers that may limit sustained engagement with digital health programs [<xref ref-type="bibr" rid="ref1">1</xref>]. Gender-diverse participants had substantially lower adjusted odds of retention (aOR 0.38), a finding that persisted across the full multivariable model. Transgender and gender-nonconforming individuals experience elevated rates of housing instability, employment discrimination, and mistrust of health and research institutions [<xref ref-type="bibr" rid="ref1">1</xref>], any of which may affect their ability to maintain participation over a 12-month study period. These findings suggest that retention support efforts should not only be tailored to recruitment platforms but also prioritize outreach to HIV-positive and gender-diverse participants, who face independently elevated risks of dropout regardless of how they were enrolled.</p><p>A sensitivity analysis using a stricter retention definition (ie, completion of all 4 follow-up surveys) was directionally consistent with the primary findings. The social media advantage over partner-seeking apps remained significant under this stricter threshold, confirming that this association is not an artifact of the chosen retention cutoff. The study contacts advantage attenuated and no longer reached significance when full completion was required, suggesting that prior study participants reengage readily but may face the same real-world barriers to sustained long-term participation as other groups. This distinction is worth considering in future studies that rely heavily on prior participant networks for recruitment.</p><p>To complement the inferential analysis, we developed a predictive logistic regression model to explore whether demographic and recruitment variables could classify retention outcomes&#x2014;a hypothesis-generating exercise motivated by the potential of ML approaches to surface patterns in behavioral health data [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. The baseline model achieved high accuracy overall (77.3%) but classified only 6.1% (2/33) of low-retention participants correctly, reflecting the limitations of standard logistic regression under class imbalance. Because the dataset was heavily skewed toward high-retention participants, the model&#x2019;s estimates for predictors of disengagement were unreliable without correction.</p><p>To address this, we applied SMOTE to synthetically balance the training dataset. The SMOTE-adjusted model achieved an accuracy of 64.7%, with low-retention recall improving from 6.1% (2/33) to 63.6% (21/33) and the <italic>F</italic><sub>1</sub>-score for that group improving from 0.11 to 0.44, at the cost of a reduction in overall accuracy. Importantly, the goal of applying SMOTE was not to predict which individual participants would drop out, but to strengthen the model&#x2019;s ability to detect group-level patterns associated with lower retention. The direction of the SMOTE model coefficients was largely consistent with the adjusted inferential analysis: social media and study contacts showed the strongest positive associations with retention, while HIV-positive status and gender-diverse identity were negatively associated. The coefficient for community-based recruitment diverged from the adjusted regression estimate, likely reflecting instability given the small subsample available for model training (n&#x2248;29 after the 80/20 split); the adjusted logistic regression provides the more reliable estimate for that comparison.</p><p>Our findings suggest that retention strategies should be tailored based on the recruitment platform&#x2019;s digital &#x201C;friction.&#x201D; For participants recruited through highly goal-oriented channels such as partner-seeking apps, informatics-based interventions could incorporate targeted engagement features to mimic the rapport seen in established research networks. Specific digital &#x201C;nudges&#x201D; might include personalized push notifications tailored to the participant&#x2019;s study milestone, milestone-based incentives, and gamified progress indicators that visually reward longitudinal participation. Additionally, incorporating peer interaction prompts or &#x201C;community boards&#x201D; may help build the institutional trust and social connectivity that characterized our high-retention study contact group. These design strategies offer a scalable way to optimize both participant diversity and sustained study engagement.</p></sec><sec id="s4-2"><title>Limitations and Future Directions</title><p>This study has several limitations. First, the race variable was collected using a multiselect item, which precluded disaggregation beyond a White versus non-White or multiracial binary classification. Given that the study specifically targets Black and Latino YBLMT, future analyses should use survey instruments that capture racial subgroup membership in a format amenable to group-level comparisons, enabling more granular examination of retention differences within the non-White population.</p><p>Although age was not independently associated with retention in the primary adjusted regression, the youngest cohort (15&#x2010;19 y, n=69) showed relatively high apparent retention, and we recognize that age may nevertheless serve as a proxy for competing life responsibilities. Factors such as full-time employment, independent housing logistics, and other adult responsibilities (which correlate with older age in this demographic) may serve as barriers to longitudinal engagement compared to younger participants who may have more discretionary time. Since these specific socioeconomic status indicators were not collected in the parent trial, we cannot definitively separate the impact of life stage from economic stability. Future research should prioritize the inclusion of comprehensive socioeconomic status metrics to better understand these overlapping influences on study persistence. Additionally, reliance on self-reported data and survey completion introduces the possibility of selection or response bias. Beyond these concerns, our study did not assess individual-level factors such as motivation, competing life responsibilities, or platform usability. Integrating qualitative data or engagement analytics could provide deeper insights into these factors.</p><p>Because this analysis was conducted using data from a parent RCT, the study was not originally powered to detect differences in retention across recruitment sources. The findings should therefore be interpreted as exploratory and hypothesis-generating. Although multiple digital platforms were used for recruitment (eg, Facebook, Instagram, Grindr, Jack&#x2019;d, Scruff), the sample size within individual platforms was insufficient to conduct reliable platform-specific comparisons. Future research examining cross-platform comparisons may be warranted [<xref ref-type="bibr" rid="ref26">26</xref>], as it may elucidate how differences in platform design, user demographics, and interaction norms influence both recruitment efficiency and longitudinal engagement in digital health interventions. Such analyses could help researchers more strategically allocate recruitment resources and tailor retention strategies to the behavioral patterns and expectations associated with specific digital environments.</p><p>Our logistic regression model successfully identified key predictors of retention, including recruitment source, age, race, gender identity, and HIV status. However, its ability to classify low-retention participants was limited, pointing to the need for refinement of predictive approaches. A sensitivity analysis using a stricter retention definition (completion of all 4 surveys) was conducted and found to be directionally consistent with the primary findings, though the study contacts advantage attenuated under this threshold.</p><p>Finally, we used SMOTE to address class imbalance in retention outcomes. While this method improved recall for low-retention participants from 6.1% (2/33) to 63.6% (21/33) and increased the <italic>F</italic><sub>1</sub>-score from 0.11 to 0.44, it reduced overall accuracy from 77.3% to 64.7%. Other resampling or model-weighting techniques may yield different trade-offs and should be explored in future work.</p></sec><sec id="s4-3"><title>Conclusions</title><p>Recruitment source was associated with participant retention in this online health intervention, highlighting its importance in digital research planning. Some platforms correspond to higher retention, while others enhance sample diversity or reach populations at increased HIV risk. As digital advertising costs rise, researchers must weigh recruitment efficiency against retention feasibility. By refining recruitment strategies and retention frameworks, future online HIV interventions can be optimized to better sustain participation among populations most impacted by the epidemic.</p></sec></sec></body><back><ack><p>The authors thank the HealthMpowerment 2.0 study participants and staff for their contributions to this research. AI tools, including Grammarly and Claude, were used during the writing process solely as editing aids to refine language and improve clarity. No AI tool was used to generate original study concepts, data, or analytical results.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the National Institute on Minority Health and Health Disparities (multiple principal investigators: JAB and KEM, R01MD013623). The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AIC</term><def><p>Akaike Information Criterion</p></def></def-item><def-item><term id="abb2">aOR</term><def><p>adjusted odds ratio</p></def></def-item><def-item><term id="abb3">CASI</term><def><p>computer-assisted self-interview</p></def></def-item><def-item><term id="abb4">HMP</term><def><p>HealthMpowerment</p></def></def-item><def-item><term id="abb5">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb6">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb7">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb8">RCT</term><def><p>randomized controlled trial</p></def></def-item><def-item><term id="abb9">SMOTE</term><def><p>synthetic minority oversampling technique</p></def></def-item><def-item><term id="abb10">USPSTF</term><def><p>United States Preventive Services Task Force</p></def></def-item><def-item><term id="abb11">YBLMT</term><def><p>Young Black and Latino men who have sex with men and transgender women who have sex with men</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arrington-Sanders</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hailey-Fair</surname><given-names>K</given-names> </name><name name-style="western"><surname>Wirtz</surname><given-names>AL</given-names> </name><etal/></person-group><article-title>Role of structural marginalization, HIV stigma, and mistrust on HIV prevention and treatment among young Black Latinx men who have sex with men and transgender women: perspectives from youth service providers</article-title><source>AIDS Patient Care STDS</source><year>2020</year><month>01</month><volume>34</volume><issue>1</issue><fpage>7</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1089/apc.2019.0165</pub-id><pub-id pub-id-type="medline">31944853</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Santos</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Beck</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>PA</given-names> </name><etal/></person-group><article-title>Homophobia as a barrier to HIV prevention service access for young men who have sex with men</article-title><source>J Acquir Immune Defic Syndr</source><year>2013</year><month>08</month><day>15</day><volume>63</volume><issue>5</issue><fpage>e167</fpage><lpage>70</lpage><pub-id pub-id-type="doi">10.1097/QAI.0b013e318294de80</pub-id><pub-id pub-id-type="medline">24135782</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Muessig</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Baltierra</surname><given-names>NB</given-names> </name><name name-style="western"><surname>Pike</surname><given-names>EC</given-names> </name><name name-style="western"><surname>LeGrand</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hightow-Weidman</surname><given-names>LB</given-names> </name></person-group><article-title>Achieving HIV risk reduction through HealthMpowerment.org, a user-driven eHealth intervention for young Black men who have sex with men and transgender women who have sex with men</article-title><source>Digit Cult Educ</source><year>2014</year><volume>6</volume><issue>3</issue><fpage>164</fpage><lpage>182</lpage><pub-id pub-id-type="medline">25593616</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hightow-Weidman</surname><given-names>LB</given-names> </name><name name-style="western"><surname>Muessig</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Bauermeister</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>C</given-names> </name><name name-style="western"><surname>LeGrand</surname><given-names>S</given-names> </name></person-group><article-title>Youth, technology, and HIV: recent advances and future directions</article-title><source>Curr HIV/AIDS Rep</source><year>2015</year><month>12</month><volume>12</volume><issue>4</issue><fpage>500</fpage><lpage>515</lpage><pub-id pub-id-type="doi">10.1007/s11904-015-0280-x</pub-id><pub-id pub-id-type="medline">26385582</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harris</surname><given-names>J</given-names> </name><name name-style="western"><surname>Germain</surname><given-names>J</given-names> </name><name name-style="western"><surname>McCoy</surname><given-names>E</given-names> </name><name name-style="western"><surname>Schofield</surname><given-names>R</given-names> </name></person-group><article-title>Ethical guidance for conducting health research with online communities: a scoping review of existing guidance</article-title><source>PLoS One</source><year>2024</year><volume>19</volume><issue>5</issue><fpage>e0302924</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0302924</pub-id><pub-id pub-id-type="medline">38758778</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guillory</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wiant</surname><given-names>KF</given-names> </name><name name-style="western"><surname>Farrelly</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Recruiting hard-to-reach populations for survey research: using Facebook and Instagram advertisements and in-person intercept in LGBT bars and nightclubs to recruit LGBT young adults</article-title><source>J Med Internet Res</source><year>2018</year><month>06</month><day>18</day><volume>20</volume><issue>6</issue><fpage>e197</fpage><pub-id pub-id-type="doi">10.2196/jmir.9461</pub-id><pub-id pub-id-type="medline">29914861</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Teitcher</surname><given-names>JEF</given-names> </name><name name-style="western"><surname>Bockting</surname><given-names>WO</given-names> </name><name name-style="western"><surname>Bauermeister</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Hoefer</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Miner</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Klitzman</surname><given-names>RL</given-names> </name></person-group><article-title>Detecting, preventing, and responding to &#x201C;fraudsters&#x201D; in internet research: ethics and tradeoffs</article-title><source>J Law Med Ethics</source><year>2015</year><volume>43</volume><issue>1</issue><fpage>116</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1111/jlme.12200</pub-id><pub-id pub-id-type="medline">25846043</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marshall</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Gower</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Katz</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Bauermeister</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Shoben</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Reiter</surname><given-names>PL</given-names> </name></person-group><article-title>Recruitment of young gay, bisexual, and other men who have sex with men for a web-based human papillomavirus vaccination intervention: differences in participant characteristics and study engagement by recruitment source in a randomized controlled trial</article-title><source>J Med Internet Res</source><year>2025</year><month>01</month><day>3</day><volume>27</volume><fpage>e64668</fpage><pub-id pub-id-type="doi">10.2196/64668</pub-id><pub-id pub-id-type="medline">39752644</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Muessig</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Golinkoff</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Hightow-Weidman</surname><given-names>LB</given-names> </name><etal/></person-group><article-title>Increasing HIV testing and viral suppression via stigma reduction in a social networking mobile health intervention among Black and Latinx young men and transgender women who have sex with men (HealthMpowerment): protocol for a randomized controlled trial</article-title><source>JMIR Res Protoc</source><year>2020</year><month>12</month><day>16</day><volume>9</volume><issue>12</issue><fpage>e24043</fpage><pub-id pub-id-type="doi">10.2196/24043</pub-id><pub-id pub-id-type="medline">33325838</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hirshfield</surname><given-names>S</given-names> </name><name name-style="western"><surname>Diaz</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Trial enrollment correlates in an HIV status-neutral mHealth intervention among young Black and Latinx men and transgender women who have sex with men</article-title><source>AIDS Behav</source><year>2026</year><month>04</month><volume>30</volume><issue>4</issue><fpage>1221</fpage><lpage>1228</lpage><pub-id pub-id-type="doi">10.1007/s10461-025-04951-0</pub-id><pub-id pub-id-type="medline">41266923</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harris</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Helfand</surname><given-names>M</given-names> </name><name name-style="western"><surname>Woolf</surname><given-names>SH</given-names> </name><etal/></person-group><article-title>Current methods of the US Preventive Services Task Force: a review of the process</article-title><source>Am J Prev Med</source><year>2001</year><month>04</month><volume>20</volume><issue>3 Suppl</issue><fpage>21</fpage><lpage>35</lpage><pub-id pub-id-type="doi">10.1016/s0749-3797(01)00261-6</pub-id><pub-id pub-id-type="medline">11306229</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chawla</surname><given-names>NV</given-names> </name><name name-style="western"><surname>Bowyer</surname><given-names>KW</given-names> </name><name name-style="western"><surname>Hall</surname><given-names>LO</given-names> </name><name name-style="western"><surname>Kegelmeyer</surname><given-names>WP</given-names> </name></person-group><article-title>SMOTE: synthetic minority over-sampling technique</article-title><source>J Artif Intell Res</source><year>2002</year><volume>16</volume><fpage>321</fpage><lpage>357</lpage><pub-id pub-id-type="doi">10.1613/jair.953</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McKinney</surname><given-names>W</given-names> </name></person-group><article-title>Data structures for statistical computing in Python</article-title><source>Proc 9th Python Sci Conf</source><year>2010</year><fpage>56</fpage><lpage>61</lpage><pub-id pub-id-type="doi">10.25080/Majora-92bf1922-00a</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pedregosa</surname><given-names>F</given-names> </name><name name-style="western"><surname>Varoquaux</surname><given-names>G</given-names> </name><name name-style="western"><surname>Gramfort</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Scikit-learn: machine learning in Python</article-title><source>J Mach Learn Res</source><year>2011</year><access-date>2026-08-06</access-date><volume>12</volume><fpage>2825</fpage><lpage>2830</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.org/papers/volume12/pedregosa11a/pedregosa11a.pdf">https://jmlr.org/papers/volume12/pedregosa11a/pedregosa11a.pdf</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lema&#x00EE;tre</surname><given-names>G</given-names> </name><name name-style="western"><surname>Nogueira</surname><given-names>F</given-names> </name><name name-style="western"><surname>Aridas</surname><given-names>CK</given-names> </name></person-group><article-title>Imbalanced-learn: a Python toolbox to tackle the curse of imbalanced datasets in machine learning</article-title><source>J Mach Learn Res</source><year>2017</year><access-date>2026-08-06</access-date><volume>18</volume><fpage>1</fpage><lpage>5</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.jmlr.org/papers/volume18/16-365/16-365.pdf">https://www.jmlr.org/papers/volume18/16-365/16-365.pdf</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seabold</surname><given-names>S</given-names> </name><name name-style="western"><surname>Perktold</surname><given-names>J</given-names> </name></person-group><article-title>Statsmodels: econometric and statistical modeling with Python</article-title><source>Proc 9th Python Sci Conf</source><year>2010</year><fpage>92</fpage><lpage>96</lpage><pub-id pub-id-type="doi">10.25080/Majora-92bf1922-011</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>J</given-names> </name><name name-style="western"><surname>Joseph</surname><given-names>K</given-names> </name></person-group><article-title>Theory in, theory out: the uses of social theory in machine learning for social science</article-title><source>Front Big Data</source><year>2020</year><volume>3</volume><fpage>18</fpage><pub-id pub-id-type="doi">10.3389/fdata.2020.00018</pub-id><pub-id pub-id-type="medline">33693392</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ellison</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Gotelli</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Inouye</surname><given-names>BD</given-names> </name><name name-style="western"><surname>Strong</surname><given-names>DR</given-names> </name></person-group><article-title>P values, hypothesis testing, and model selection: it&#x2019;s d&#x00E9;j&#x00E0; vu all over again</article-title><source>Ecology</source><year>2014</year><month>03</month><volume>95</volume><issue>3</issue><fpage>609</fpage><lpage>610</lpage><pub-id pub-id-type="doi">10.1890/13-1911.1</pub-id><pub-id pub-id-type="medline">24804440</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>G&#x00E9;ron</surname><given-names>A</given-names></name></person-group><source>Hands-On Machine Learning with Scikit-Learn, Keras, and TensorFlow</source><year>2022</year><access-date>2025-04-01</access-date><edition>3</edition><publisher-name>O&#x2019;Reilly Media</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.oreilly.com/library/view/hands-on-machine-learning/9781492032632/">https://www.oreilly.com/library/view/hands-on-machine-learning/9781492032632/</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Buda</surname><given-names>M</given-names> </name><name name-style="western"><surname>Maki</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mazurowski</surname><given-names>MA</given-names> </name></person-group><article-title>A systematic study of the class imbalance problem in convolutional neural networks</article-title><source>Neural Netw</source><year>2018</year><month>10</month><volume>106</volume><fpage>249</fpage><lpage>259</lpage><pub-id pub-id-type="doi">10.1016/j.neunet.2018.07.011</pub-id><pub-id pub-id-type="medline">30092410</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Whitaker</surname><given-names>C</given-names> </name><name name-style="western"><surname>Stevelink</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fear</surname><given-names>N</given-names> </name></person-group><article-title>The use of Facebook in recruiting participants for health research purposes: a systematic review</article-title><source>J Med Internet Res</source><year>2017</year><month>08</month><day>28</day><volume>19</volume><issue>8</issue><fpage>e290</fpage><pub-id pub-id-type="doi">10.2196/jmir.7071</pub-id><pub-id pub-id-type="medline">28851679</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Topolovec-Vranic</surname><given-names>J</given-names> </name><name name-style="western"><surname>Natarajan</surname><given-names>K</given-names> </name></person-group><article-title>The use of social media in recruitment for medical research studies: a scoping review</article-title><source>J Med Internet Res</source><year>2016</year><month>11</month><day>7</day><volume>18</volume><issue>11</issue><fpage>e286</fpage><pub-id pub-id-type="doi">10.2196/jmir.5698</pub-id><pub-id pub-id-type="medline">27821383</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Booker</surname><given-names>CL</given-names> </name><name name-style="western"><surname>Harding</surname><given-names>S</given-names> </name><name name-style="western"><surname>Benzeval</surname><given-names>M</given-names> </name></person-group><article-title>A systematic review of the effect of retention methods in population-based cohort studies</article-title><source>BMC Public Health</source><year>2011</year><month>04</month><day>19</day><volume>11</volume><issue>1</issue><fpage>249</fpage><pub-id pub-id-type="doi">10.1186/1471-2458-11-249</pub-id><pub-id pub-id-type="medline">21504610</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Galea</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tracy</surname><given-names>M</given-names> </name></person-group><article-title>Participation rates in epidemiologic studies</article-title><source>Ann Epidemiol</source><year>2007</year><month>09</month><volume>17</volume><issue>9</issue><fpage>643</fpage><lpage>653</lpage><pub-id pub-id-type="doi">10.1016/j.annepidem.2007.03.013</pub-id><pub-id pub-id-type="medline">17553702</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Chan</surname><given-names>LS</given-names> </name></person-group><source>The Politics of Dating Apps: Gender, Sexuality, and Emergent Publics in Urban China</source><year>2021</year><publisher-name>The MIT Press</publisher-name><pub-id pub-id-type="doi">10.7551/mitpress/12742.001.0001</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zlotorzynska</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bauermeister</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Golinkoff</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>W</given-names> </name><name name-style="western"><surname>Sanchez</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Hightow-Weidman</surname><given-names>L</given-names> </name></person-group><article-title>Online recruitment of youth for mHealth studies</article-title><source>mHealth</source><year>2021</year><volume>7</volume><fpage>27</fpage><pub-id pub-id-type="doi">10.21037/mhealth-20-64</pub-id><pub-id pub-id-type="medline">33898596</pub-id></nlm-citation></ref></ref-list></back></article>