<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="letter"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e93703</article-id><article-id pub-id-type="doi">10.2196/93703</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Letter</subject></subj-group></article-categories><title-group><article-title>Enhancing Data Integrity in Online Health Surveys Through Multilayered Security Measures: Cross-Sectional Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Mora Pinzon</surname><given-names>Maria</given-names></name><degrees>MS, MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Fernandez de Cordova</surname><given-names>Susana</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Lor</surname><given-names>Maichou</given-names></name><degrees>RN, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Medicine, Division of Geriatrics and Gerontology, School of Medicine and Public Health, University of Wisconsin&#x2013;Madison</institution><addr-line>610 Walnut St</addr-line><addr-line>Madison</addr-line><addr-line>WI</addr-line><country>United States</country></aff><aff id="aff2"><institution>School of Nursing, University of Wisconsin&#x2013;Madison</institution><addr-line>Madison</addr-line><addr-line>WI</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Mavragani</surname><given-names>Amaryllis</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Guy</surname><given-names>Arryn A</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Sinamo</surname><given-names>Joshua K</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Maria Mora Pinzon, MS, MD, Department of Medicine, Division of Geriatrics and Gerontology, School of Medicine and Public Health, University of Wisconsin&#x2013;Madison, 610 Walnut St, Madison, WI, 53726, United States, 1 6088902524; <email>mcpinzon@medicine.wisc.edu</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>all authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>2</day><month>9</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e93703</elocation-id><history><date date-type="received"><day>17</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>11</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>17</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Maria Mora Pinzon, Susana Fernandez de Cordova, Maichou Lor. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 2.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e93703"/><abstract><p>A multilayered fraud-mitigation approach is essential to ensure data integrity in medical survey research; basic measures alone (eg, CAPTCHA) would permit widespread fraud.</p></abstract><kwd-group><kwd>online surveys</kwd><kwd>bots</kwd><kwd>fraud</kwd><kwd>verification</kwd><kwd>security measures</kwd><kwd>data integrity</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Online health surveys have become essential tools for public health and clinical research, yet they are increasingly vulnerable to sophisticated fraudulent responses, impacting data integrity and quality. Fraudulent responses are submissions that contain false, fabricated, or misleading information provided to misrepresent participation eligibility and that may include automated entries from bots or repeated entries by the same actor [<xref ref-type="bibr" rid="ref1">1</xref>]. Pinz&#x00F3;n et&#x202F;al [<xref ref-type="bibr" rid="ref2">2</xref>] documented a dramatic erosion of usable data over time, underscoring how AI-enabled fraud and incentive-seeking behavior can overwhelm open links. Similar issues have been reported in nursing and medical research, where 70% to 94% or more of survey responses were identified as fraudulent or automated through bots [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref6">6</xref>]. These findings suggest that traditional safeguards, such as CAPTCHA, are insufficient for data integrity [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. This study&#x2019;s purpose is to describe the results of implementing a multilayered security approach that integrates automated and human-review mechanisms in a nationwide survey of health care professionals.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>We conducted a cross-sectional online survey of health care professionals across the United States between November 2024 and May 2025. Clinicians aged 18 years or older practicing in the United States who have worked with patients with limited English proficiency were eligible to participate. The survey examined pain-related communication challenges with Spanish-speaking patients with limited English proficiency.</p><p>Recruitment occurred through medical professional organization newsletters and websites, health care&#x2013;focused Facebook and WhatsApp (Meta Platforms Inc) groups, and the authors&#x2019; professional Facebook and Instagram (Meta Platforms Inc) pages. No paid advertising was used. We used a 2-tiered process, where posts included a direct link or QR code to the screening survey and then eligible respondents received an individualized link to a Qualtrics (Qualtrics LLC) survey.</p><p>Our multilayered security approach combines automated fraud indicators (<xref ref-type="table" rid="table1">Table 1</xref>) with a human verification process. Some indicator thresholds were adapted from prior literature based on patterns observed in our previous fraud detection work and in the current dataset. Using R software (version 4.5.1; R Foundation for Statistical Computing), we developed a fraud detection workflow that followed these steps: (1) entries were automatically disqualified based on eligibility criteria (eg, non&#x2013;health care providers) and definitive fraud indicators (eg, non-US submissions), (2) remaining submissions underwent manual review where entries with more than 2 indicators received brief verification of fraudulent status, and (3) entries with 1 or 2 indicators underwent human manual verification, including confirming provider credentials using National Provider Identifier records or state professional licensing board databases. No submission was classified as fraudulent based on a single nondefinitive indicator.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Comprehensive fraud indicators with definitions [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>].</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Indicator name</td><td align="left" valign="bottom">Definition</td><td align="left" valign="bottom">Threshold or activity considered fraudulent</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Technical indicators</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qualtrics (Qualtrics LLC) RelevantID fraud score</td><td align="left" valign="top">Qualtrics&#x2019; proprietary metric, measures the likelihood of fraudulent response based on multiple factors, such as how respondents interact with the survey and the metadata they generate, to spot signs of fraud or abuse. This tool was discontinued in June 2025.</td><td align="left" valign="top">Score of &#x2265;30</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Non-US location</td><td align="left" valign="top">Location of submission identified as outside United States based on IP<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> address.</td><td align="left" valign="top">Location outside United States</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High duplicate score</td><td align="left" valign="top">Qualtrics&#x2019;s built-in metric, measures similarity between survey responses to detect potential duplicates.</td><td align="left" valign="top">Score of &#x2265;75</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Duplicate name</td><td align="left" valign="top">Exact name match in 2 or more responses with different accompanying information.</td><td align="left" valign="top">&#x2265;2 occurrences</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Duplicate email</td><td align="left" valign="top">Exact email match in 2 or more responses with different accompanying information.</td><td align="left" valign="top">&#x2265;2 occurrences</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Duplicate IP</td><td align="left" valign="top">IP address that matches in 2 or more responses.</td><td align="left" valign="top">&#x2265;2 occurrences</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>IP block</td><td align="left" valign="top">Submission from known IP blocks associated with VPNs<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> or proxy services&#x2014;suspicious when combined with other indicators.</td><td align="left" valign="top">1 or 2 blocks</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Disposable email</td><td align="left" valign="top">Sequential submissions from same host or suspicious email domains. Temporary email services act as strong fraud indicators [<xref ref-type="bibr" rid="ref8">8</xref>].</td><td align="left" valign="top">List of disposable email providers frequently associated with fraudulent activity</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Invalid email</td><td align="left" valign="top">Found by format validation.</td><td align="left" valign="top">No @ symbol</td></tr><tr><td align="left" valign="top" colspan="3">Temporal indicators</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Regular waves</td><td align="left" valign="top">Submissions occurring in waves or at regular intervals (eg, 3 responses every 5 min), especially when batches of responses are submitted within 60 seconds at consistent intervals.</td><td align="left" valign="top">&#x2265;3 submissions per batch, &#x2265;3 intervals</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rapid submission</td><td align="left" valign="top">Patterns of responses arriving in batches, often dozens per minute, particularly overnight.</td><td align="left" valign="top">&#x2265;3 in the same minute</td></tr><tr><td align="left" valign="top" colspan="3">Email pattern indicators</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Many consecutive consonants</td><td align="left" valign="top">Character sequence with 4 or more consecutive consonants. This pattern rarely occurs in authentic email addresses, suggesting random character generation.</td><td align="left" valign="top">&#x2265;4 consecutive consonants</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Many transitions</td><td align="left" valign="top">Alternating letters and numbers in an email address, which indicates algorithmic construction (eg, a12bcd34e@email.com).</td><td align="left" valign="top">&#x2265;2 transitions</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Has long number</td><td align="left" valign="top">Addresses with 4 or more digits, which previous work found may indicate automated generation rather than meaningful numbers such as graduation years [<xref ref-type="bibr" rid="ref9">9</xref>]. Based on patterns observed in our previous work and in the current dataset, we used a threshold of 3 or more digits as a screening indicator requiring further review.</td><td align="left" valign="top">&#x2265;3 digits</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Low vowel/consonant ratio</td><td align="left" valign="top">A mathematical approach to detect unnatural letter distribution, which emerged from our pattern analysis of confirmed fraudulent addresses.</td><td align="left" valign="top">ratio of &#x2264;0.257</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Capitalized names</td><td align="left" valign="top">In the current dataset, structures like &#x201C;JaneDoe&#x201D; appeared exclusively in fraudulent responses, while legitimate respondents either used lowercase formatting or included proper punctuation (eg, &#x201C;Jane.Doe&#x201D;). This consistent finding likely results from spam generation software attempting to mimic authentic names without understanding professional email conventions [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</td><td align="left" valign="top">JaneDoe format</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Long local part</td><td align="left" valign="top">A long email handle; previous work found that no legitimate respondents had email handles exceeding 22 characters [<xref ref-type="bibr" rid="ref2">2</xref>].</td><td align="left" valign="top">&#x2265;18 characters</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Starts with number</td><td align="left" valign="top">First character is a number.</td><td align="left" valign="top">Numeric start</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No vowels in the email</td><td align="left" valign="top">An extreme deviation from natural language patterns, which emerged from our own analysis as the strongest single indicator of fraudulent generation.</td><td align="left" valign="top">0 vowels</td></tr><tr><td align="left" valign="top" colspan="3">Content indicators</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Honeypot</td><td align="left" valign="top">Interaction with hidden form fields designed to catch automated submission. Survey respondents see a question that says &#x201C;leave blank&#x201D; or a variation of it.</td><td align="left" valign="top">Not empty</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No contact information</td><td align="left" valign="top">Contact information for subsequent emails is empty.</td><td align="left" valign="top">Missing/empty</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>IP: internet protocol.</p></fn><fn id="table1fn2"><p><sup>b</sup>VPN: virtual private network.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study was approved by the University of Wisconsin-Madison&#x2019;s institutional review board (ID: 2024&#x2010;1314). Informed consent was displayed in the introduction of the survey, and all participants had to agree to participate to continue. Survey responses were stored separately from contact information to ensure privacy and confidentiality. Participants received an electronic gift card of US $15 for completing the survey.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>Our multilayered security approach identified 5846 of 5886 entries (99.32%) as potentially fraudulent, while only 40 were verified to be legitimate. The automated screening phase disqualified 2895 (49.18%) submissions. The remaining 2991 (50.82%) submissions underwent manual review. Of these, 2224 (74.35%) contained 2 or more indicators of fraud and required only brief verification, while 767 (25.64%) required more extensive examination. <xref ref-type="table" rid="table2">Table 2</xref> shows that the most frequent fraud signal was the temporal regular waves pattern (n=4447, 75.55%), followed by a high fraud score (n=1614, 27.44%) and duplicate internet protocol (n=1614, 24.88%). Combined attacks were also notable, with the regular wave &#x202F;+ suspicious email pattern being present in 418 (7.1%) responses (<xref ref-type="table" rid="table2">Table 2</xref>).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Fraud indicators identified in the survey responses between November 2024 and May 2025 (N=5886).</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Indicator name</td><td align="left" valign="bottom">Count, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Technical indicators</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High fraud score</td><td align="left" valign="top">1614 (27.44)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Duplicate IP<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top">1464 (24.88)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>IP block</td><td align="left" valign="top">860 (14.61)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Non-US location</td><td align="left" valign="top">733 (12.45)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High duplicate score</td><td align="left" valign="top">606 (10.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Duplicate name</td><td align="left" valign="top">177 (3.01)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Duplicate email</td><td align="left" valign="top">168 (2.85)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Disposable email</td><td align="left" valign="top">74 (1.26)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Invalid email</td><td align="left" valign="top">5 (0.08)</td></tr><tr><td align="left" valign="top">Temporal indicators</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Regular waves</td><td align="left" valign="top">4447 (75.55)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rapid submission</td><td align="left" valign="top">1311 (22.27)</td></tr><tr><td align="left" valign="top">Email pattern indicators</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Long number</td><td align="left" valign="top">1261 (21.43)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Many consecutive consonants</td><td align="left" valign="top">935 (15.89)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Capitalized names</td><td align="left" valign="top">589 (10.01)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Low vowel/consonant ratio</td><td align="left" valign="top">302 (5.13)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Long local part</td><td align="left" valign="top">209 (3.55)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No vowels</td><td align="left" valign="top">202 (3.43)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Many transitions</td><td align="left" valign="top">142 (2.41)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Starts with number</td><td align="left" valign="top">21 (0.36)</td></tr><tr><td align="left" valign="top">Content indicators</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No contact information</td><td align="left" valign="top">80 (1.36)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Honeypot</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">Combined attacks</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Regular wave pattern + suspicious email pattern</td><td align="left" valign="top">418 (7.1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High fraud score + regular wave pattern</td><td align="left" valign="top">279 (4.74)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rapid submission + regular wave pattern + suspicious email pattern</td><td align="left" valign="top">279 (4.74)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rapid submission + regular wave pattern</td><td align="left" valign="top">209 (3.55)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Regular wave pattern + email has 3 or more consecutive digits</td><td align="left" valign="top">201 (3.41)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>IP: internet protocol.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This multilayered security approach proved highly effective at identifying submissions with multiple indicators of fraud and distinguishing them from submissions from verified health care professionals through credential verification and manual review. Among individual signals of fraud, the most discriminative was the temporal regular wave pattern. By contrast, the honeypot captured no entries, and submissions with no contact information were rare (1.4%) and nonspecific. This suggests that contemporary threats originating from sophisticated automated systems are capable of avoiding traditional bot-detection mechanisms.</p><p>No single indicator was sufficient for adjudication because legitimate circumstances could generate these signals. For example, regular wave patterns may result from social media distribution, duplicate IP addresses may occur when using shared networks, and character-based indicators (eg, a low vowel/consonant ratio) may reflect legitimate naming conventions (eg, Schmidt). Thus, human review is essential to minimize false positives when scaling verification.</p><p>Our findings confirm that no single indicator is sufficient; detection improves when multiple indicators are combined [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Guy et al [<xref ref-type="bibr" rid="ref10">10</xref>] reported similar results in 2 online studies of marginalized populations, showing that reliance on a single fraud detection approach would have resulted in the majority of bots or fraudulent responses remaining undetected. In our data, even the strongest single signal would have missed nearly one-quarter of fraud if used alone. Multilayered defenses enabled the accurate removal of 16% to 88% of fraudulent entries depending on context; where permissible, personal identifiers would further improve yield [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>Several limitations should be acknowledged. First, we cannot formally estimate sensitivity, specificity, or overall classification accuracy, including naming traditions, because there is no gold standard to determine each submission&#x2019;s true status. Second, all recruitment channels used the same screening survey; therefore, we could not compare fraud rates across channels.</p><p>As online survey fraud continues to evolve, researchers must remain vigilant and adaptive. The extreme fraud rate documented here may reflect a new reality for online studies, particularly those offering monetary incentives. While resource intensive, our results demonstrate that comprehensive fraud detection can successfully preserve data integrity even in heavily targeted surveys. Our findings also highlight the importance of recruitment strategies. Public social media posts through professional organizations and health care&#x2013;focused groups may increase survey visibility and simultaneously increase exposure to fraud. Future studies should evaluate whether recruitment through professional organizations&#x2019; members-only platforms reduces fraud while maintaining adequate reach.</p></sec><sec id="s4-2"><title>Conclusions</title><p>This study demonstrates that safeguarding online health survey data requires a comprehensive multitiered framework. As technology evolves, adaptive approaches will be essential to maintain the validity, credibility, and reproducibility of web-based research. Furthermore, we provide practical insights into recruiting health care professionals through professional organizations and public social media channels. Investigators should anticipate recruitment inefficiencies, and they may need to cast a wider recruitment net than anticipated and allocate substantial resources to response verification.</p></sec></sec></body><back><ack><p>We thank George Levy and Maria Rosales for their work in the early stages of this study.</p><p>We acknowledge the use of Microsoft Copilot (version 4.0; OpenAI and Microsoft) and Grammarly (version 1.5; Grammarly Inc.) during the preparation of this manuscript. The AI tools were used to assist in the creation, review, and revision of the content. Specifically, Copilot was used to provide suggestions for revisions and ensure clarity and coherence in the text, while Grammarly was used to enhance grammar, punctuation, and overall writing quality. The authors take full responsibility for the integrity and accuracy of the content generated by these AI tools.</p></ack><notes><sec><title>Funding</title><p>Research reported in this publication was supported by the National Institute On Aging of the National Institutes of Health under award number R00AG076966 (principal investigator MMP) and a University of Wisconsin-Madison Institute for Clinical and Translational Research voucher (principal investigators ML and MMP). The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health.</p></sec><sec><title>Data Availability</title><p>The data underlying this study are available from the authors upon reasonable request contingent upon execution of a data use agreement with the University of Wisconsin-Madison.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">IP</term><def><p>internet protocol</p></def></def-item><def-item><term id="abb2">VPN</term><def><p>virtual privacy network</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>King Stokes</surname><given-names>N</given-names> </name><name name-style="western"><surname>McCusker</surname><given-names>S</given-names> </name><name name-style="western"><surname>Daly</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Fraudulent responses in online survey research in dermatology: are your participants who you think they are?</article-title><source>Clin Exp Dermatol</source><year>2026</year><month>03</month><day>26</day><volume>51</volume><issue>4</issue><fpage>626</fpage><lpage>629</lpage><pub-id pub-id-type="doi">10.1093/ced/llaf517</pub-id><pub-id pub-id-type="medline">41271214</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pinz&#x00F3;n</surname><given-names>N</given-names> </name><name name-style="western"><surname>Koundinya</surname><given-names>V</given-names> </name><name name-style="western"><surname>Galt</surname><given-names>RE</given-names> </name><etal/></person-group><article-title>AI-powered fraud and the erosion of online survey integrity: an analysis of 31 fraud detection strategies</article-title><source>Front Res Metr Anal</source><year>2024</year><volume>9</volume><fpage>1432774</fpage><pub-id pub-id-type="doi">10.3389/frma.2024.1432774</pub-id><pub-id pub-id-type="medline">39687573</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Matos</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Silva</surname><given-names>S</given-names> </name><name name-style="western"><surname>Relf</surname><given-names>MV</given-names> </name><name name-style="western"><surname>Gonzalez-Guarda</surname><given-names>R</given-names> </name></person-group><article-title>Addressing survey fraud in online health research: a case study of Latine sexual minority men</article-title><source>Res Nurs Health</source><year>2025</year><month>12</month><volume>48</volume><issue>6</issue><fpage>750</fpage><lpage>762</lpage><pub-id pub-id-type="doi">10.1002/nur.70021</pub-id><pub-id pub-id-type="medline">41001781</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nur</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Leibbrand</surname><given-names>C</given-names> </name><name name-style="western"><surname>Curran</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Votruba-Drzal</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gibson-Davis</surname><given-names>C</given-names> </name></person-group><article-title>Managing and minimizing online survey questionnaire fraud: Lessons from the Triple C project</article-title><source>Int J Soc Res Methodol</source><year>2024</year><volume>27</volume><issue>5</issue><fpage>613</fpage><lpage>619</lpage><pub-id pub-id-type="doi">10.1080/13645579.2023.2229651</pub-id><pub-id pub-id-type="medline">39494158</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schles</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Deheck</surname><given-names>CM</given-names> </name></person-group><article-title>Identifying and mitigating the influence of invalid responses in online surveys: a longitudinal case study</article-title><source>Qual Quant</source><year>2025</year><volume>60</volume><issue>1</issue><fpage>1903</fpage><lpage>1921</lpage><pub-id pub-id-type="doi">10.1007/s11135-025-02333-1</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pozzar</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hammer</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Underhill-Blazey</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Threats of bots and other bad actors to data quality following research participant recruitment through social media: cross-sectional questionnaire</article-title><source>J Med Internet Res</source><year>2020</year><month>10</month><day>7</day><volume>22</volume><issue>10</issue><fpage>e23021</fpage><pub-id pub-id-type="doi">10.2196/23021</pub-id><pub-id pub-id-type="medline">33026360</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ennis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Renner</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Morando-Stokoe</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Exploring methods to mitigate fraud in web-based surveys: multicase study analysis</article-title><source>J Med Internet Res</source><year>2025</year><month>12</month><day>1</day><volume>27</volume><fpage>e78671</fpage><pub-id pub-id-type="doi">10.2196/78671</pub-id><pub-id pub-id-type="medline">41324984</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Storozuk</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ashley</surname><given-names>M</given-names> </name><name name-style="western"><surname>Delage</surname><given-names>V</given-names> </name><name name-style="western"><surname>Maloney</surname><given-names>EA</given-names> </name></person-group><article-title>Got bots? Practical recommendations to protect online survey data from bot attacks</article-title><source>Quant Meth Psych</source><year>2020</year><volume>16</volume><issue>5</issue><fpage>472</fpage><lpage>481</lpage><pub-id pub-id-type="doi">10.20982/tqmp.16.5.p472</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Griffin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Martino</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>LoSchiavo</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Ensuring survey research data integrity in the era of internet bots</article-title><source>Qual Quant</source><year>2022</year><volume>56</volume><issue>4</issue><fpage>2841</fpage><lpage>2852</lpage><pub-id pub-id-type="doi">10.1007/s11135-021-01252-1</pub-id><pub-id pub-id-type="medline">34629553</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guy</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Zelaya</surname><given-names>DG</given-names> </name><name name-style="western"><surname>Kahler</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>S</given-names> </name></person-group><article-title>Data integrity in an online world: demonstration of multimodal bot screening tools and considerations for preserving data integrity in two online social and behavioral research studies with marginalized populations</article-title><source>Psychol Methods</source><year>2024</year><month>09</month><day>9</day><volume>doi</volume><pub-id pub-id-type="doi">10.1037/met0000696</pub-id><pub-id pub-id-type="medline">39250292</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dewitt</surname><given-names>J</given-names> </name><name name-style="western"><surname>Capistrant</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kohli</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Addressing participant validity in a small internet health survey (the Restore Study): protocol and recommendations for survey response validation</article-title><source>JMIR Res Protoc</source><year>2018</year><month>04</month><day>24</day><volume>7</volume><issue>4</issue><fpage>e96</fpage><pub-id pub-id-type="doi">10.2196/resprot.7655</pub-id><pub-id pub-id-type="medline">29691203</pub-id></nlm-citation></ref></ref-list></back></article>