<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e88549</article-id><article-id pub-id-type="doi">10.2196/88549</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Dual-Source Retrieval-Augmented Generation Chatbot for Women&#x2019;s Health (HerCare): Design and Multimethod Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Zaman</surname><given-names>Kimia Tuz</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hasan</surname><given-names>Wordh Ul</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ahmed</surname><given-names>Nova</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Li</surname><given-names>Juan</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Computer Science Department, North Dakota State University</institution><addr-line>258 Quentin Burdick Building NDSU, 1320 Albrecht Boulevard</addr-line><addr-line>Fargo</addr-line><addr-line>ND</addr-line><country>United States</country></aff><aff id="aff2"><institution>Tuskegee University</institution><addr-line>Tuskegee</addr-line><addr-line>AL</addr-line><country>United States</country></aff><aff id="aff3"><institution>North South University</institution><addr-line>Dhaka</addr-line><addr-line>Dhaka Division</addr-line><country>Bangladesh</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Law</surname><given-names>Stephanie</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Vorisek</surname><given-names>Carina</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Babalola</surname><given-names>Wuraola Susan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Juan Li, PhD, Computer Science Department, North Dakota State University, 258 Quentin Burdick Building NDSU, 1320 Albrecht Boulevard, Fargo, ND, 58105, United States, 1 701-231-9662; <email>j.li@ndsu.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>31</day><month>7</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e88549</elocation-id><history><date date-type="received"><day>27</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>17</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>17</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Kimia Tuz Zaman,Wordh Ul Hasan, Nova Ahmed, Juan Li. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 31.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e88549"/><abstract><sec><title>Background</title><p>Conversational agents for women&#x2019;s health often fail to meet user needs, offering either clinically sterile advice or unreliable peer anecdotes. This limitation creates a tension between the need for factual safety and emotional resonance in sensitive health contexts.</p></sec><sec><title>Objective</title><p>We aimed to address this gap by developing and conducting a formative evaluation of HerCare, a conversational agent built on a novel dual-source retrieval-augmented generation architecture. The system integrates expert medical knowledge with peer narratives and makes the provenance of each response visible to users, enabling trust calibration through transparent source attribution.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a remote, web-based single-session field study (December 2024 to January 2025; North Dakota State University Institutional Review Board Protocol #IRB0005368) with 243 completers (from 335 eligible, consenting visitors) recruited via social media (Facebook [Meta], Reddit, and Instagram [Meta]) and university mailing lists. Eligible participants self-identified as women aged 18&#x2010;45 years with English proficiency and internet access. We used a quantitative multimethod evaluation, combining standardized self-report metrics&#x2014;the Chatbot Usability Questionnaire and net promoter score (NPS)&#x2014;with computational linguistic analyses (VADER [Valence Aware Dictionary and Sentiment Reasoner] sentiment analysis and NRC [National Research Council] Emotion Lexicon) of 1191 conversational turns.</p></sec><sec sec-type="results"><title>Results</title><p>Among the 243 participants who completed the protocol, reported usability was high (Chatbot Usability Questionnaire median 78.1, IQR 65.2&#x2010;87.5; mean 75.67, SD 15.50) and advocacy was strong (NPS 60.0; 171/243, 70.4% promoters, 25/243, 10.3% detractors), though this NPS reflects completers only. Postinteraction ratings were high (all facets median 4&#x2010;5 on a 5-point scale; helpfulness, ease of use, and clarity median 5, IQR 4-5). Computational analysis revealed a consistent polarity shift from neutral to negative user queries (compound &#x2212;0.18 to +0.15) to strongly positive agent responses (compound +0.55 to +0.83), with a recurring validate-then-redirect empathy pattern in which the agent acknowledges user distress before pivoting to constructive guidance.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Among completers, the dual-source architecture was associated with high perceived empathy and trust, suggesting it can combine clinical accuracy with emotional support. These formative findings indicate the feasibility of weaving clinical sources with lived experiences toward safer, more resonant health AI and surface a candidate design pattern for future empathy-attuned systems that warrants controlled evaluation.</p></sec></abstract><kwd-group><kwd>women&#x2019;s health</kwd><kwd>conversational agents</kwd><kwd>large language models</kwd><kwd>retrieval-augmented generation</kwd><kwd>empathy</kwd><kwd>human-computer interaction</kwd><kwd>trust</kwd><kwd>usability</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>People increasingly turn to digital tools when navigating questions about fertility, contraception, periods, pregnancy, and menopause [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref5">5</xref>]. In these moments, they often seek not only accurate information but also language and guidance that feel respectful, reassuring, and aligned with their values [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. Prior human-computer interaction (HCI) and public-health work show that women&#x2019;s health experiences are shaped by emotion, culture, and trust&#x2014;factors that influence whether information feels usable and safe in practice [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. Yet, many systems are optimized to deliver facts quickly, filling information gaps while missing the care and empathy that could accompany the answer.</p><p>The central design challenge is not to choose between expert knowledge and lived experience, but to responsibly integrate them. Medical accuracy and safety demand grounding in vetted clinical sources; women&#x2019;s health journeys are also textured by peer narratives, cultural context, and emotion that purely clinical responses cannot capture. Current systems tend to treat these domains as mutually exclusive: clinical chatbots prioritize factual correctness but strip away human resonance, while community forums provide emotional solidarity but risk misinformation [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. What is missing is a principled way to weave these knowledge forms together so people do not have to choose between feeling understood and being accurately informed [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>].</p><p>Retrieval-augmented generation (RAG) has emerged as a technically promising approach to grounding conversational health AI in verifiable knowledge. However, current implementations remain narrowly conceived in their knowledge sourcing. A 2025 scoping review of 67 RAG-based health studies found that 54% drew from a single knowledge source, with the large majority retrieving exclusively from clinical literature, medical guidelines, or authoritative databases [<xref ref-type="bibr" rid="ref26">26</xref>]. Broader reviews of health care RAG systems confirm this pattern: such systems overwhelmingly retrieve from &#x201C;trusted medical knowledge bases and clinical literature&#x201D; [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>], treating expert clinical content as the sole legitimate input. Community-generated health knowledge, despite representing the primary information source for many patients navigating chronic, reproductive, or stigmatized health conditions, has been systematically excluded from RAG architectures. This single-source design imposes a ceiling on empathic performance: clinically authoritative content may ground responses factually but cannot supply the experiential validation, shared language, and emotional solidarity that users of peer health communities report as distinctively valuable.</p><p>The empathy and personalization shortfall in women&#x2019;s health conversational agents (CAs) is well documented. Qualitative work with pregnant users of a maternal chatbot identified limited personalization and the absence of empathetic interaction as primary weaknesses [<xref ref-type="bibr" rid="ref29">29</xref>], and a realist synthesis of sexual and reproductive health chatbots found that systems felt less valuable when conversations were unnatural, prompting calls to prioritize authentic conversational tone and longer-term relationship building [<xref ref-type="bibr" rid="ref30">30</xref>]. The pattern extends to perinatal and postpartum contexts: a randomized trial of a postpartum chatbot reported high satisfaction with informational content but markedly lower therapeutic-alliance scores, exposing a gap between informational adequacy and empathic presence [<xref ref-type="bibr" rid="ref31">31</xref>]. Across these evaluations, systems optimize for accurate information while underdelivering the emotional resonance women seek in sensitive health contexts.</p><p>Taken together, this evidence defines a precise and unoccupied design space. Women&#x2019;s health chatbots consistently underperform on empathy and personalization. Health RAG systems retrieve from single, clinical-only sources. Peer health knowledge is valuable but requires expert grounding to be safe. Additionally, source attribution, when architecturally enforced rather than superficially applied, measurably increases trust. No existing system has simultaneously addressed all four of these gaps at once: a single dual-source retrieval architecture that fuses vetted clinical knowledge with community peer narratives, enforces transparent source attribution at the response level, and tunes conversational tone to the user&#x2019;s affective state. HerCare is designed to occupy this gap.</p><p>HerCare is a web-based CA purpose-built for women&#x2019;s health support. The system combines curated clinical resources with peer narratives in a dual-source retrieval architecture and explicitly attributes which parts of a reply come from medical guidance vs community perspectives. An empathy-mapping layer tunes tone to the user&#x2019;s expressed needs. While further clinical validation will be required, we use HerCare to examine whether dual-source retrieval and explicit attribution can enable safe, resonant guidance, offering groundwork for future designs in empathy-attuned health AI.</p><p>The aim of this paper is twofold: first, to describe the design and development of HerCare, a CA built on a novel dual-source RAG architecture that systematically integrates expert clinical knowledge with peer community narratives under a trust-calibration framework; and second, to report a formative evaluation of this system conducted with 243 participants in a remote, single-session field study, assessing its usability, trustworthiness, and affective dynamics through standardized self-report instruments and corpus-level computational linguistic analysis.</p><p>This work makes three contributions: (1) technical feasibility of a dual-source RAG architecture that fuses vetted clinical knowledge with peer narratives under explicit attribution; (2) a multimethod quantitative framework pairing self-report instruments with corpus-level linguistic analysis to assess affective dynamics; and (3) formative feasibility evidence characterizing the validate-then-redirect empathy pattern as a transferable design template for supportive health AI.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This section is organized in two parts: system design and architecture, followed by the formative evaluation study design and procedures.</p></sec><sec id="s2-2"><title>Part 1: System Design and Development</title><sec id="s2-2-1"><title>HerCare System Overview and Architecture</title><p>HerCare is a web-based CA purpose-built for women&#x2019;s health support. The system is built on a dual-source RAG architecture that integrates two complementary knowledge sources&#x2014;vetted clinical content from the Mayo Clinic and peer narratives from Reddit women&#x2019;s health communities&#x2014;and explicitly attributes each piece of information to its origin. This transparent attribution is the system&#x2019;s core trust-calibration mechanism: by labeling whether a response draws on community experience or medical guidance, the architecture allows users to evaluate and calibrate their trust rather than accepting outputs uncritically. An empathy-mapping layer further tunes the system&#x2019;s tone to the user&#x2019;s expressed emotional state before each response is generated. An overview of the complete technical software stack and deployment infrastructure is summarized in Section S6 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref><bold>.</bold></p><p>The system operates through a four-stage pipeline: (1) query analysis and affective state inference, (2) parallel semantic retrieval from both vector stores, (3) dynamic prompt construction integrating the retrieved context, empathy goals, and grounding rules, and (4) response generation via GPT-4 with mandatory source attribution and a safety disclaimer. Each stage is described in the sections that follow. The complete architecture is summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>System diagram for multistaged user query. This four-stage pipeline illustrates how HerCare processes each user interaction from query analysis through parallel dual-source retrieval to empathy-informed response generation, demonstrating the architectural integration of peer and clinical knowledge sources.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88549_fig01.png"/></fig></sec><sec id="s2-2-2"><title>Knowledge Base Construction</title><p>The foundation of the system&#x2019;s retrieval capability is a hybrid knowledge base composed of two distinct and purpose-built corpora. The construction of this knowledge base required the development of asymmetric data processing pipelines, where ingestion, cleaning, and chunking methodologies were specifically tailored to the unique structural and semantic characteristics of each data source. This tailored approach is critical for optimizing the quality and relevance of the information available for retrieval.</p><p>To capture the authentic voice of lived experience in women&#x2019;s health, we drew our first corpus from Reddit, a social platform organized into topic-specific subreddits and shaped by pseudonymous participation. Reddit&#x2019;s design encourages frank discussion of sensitive topics and produces threaded, community-moderated dialogues that are well-suited to studying support-seeking in situ. We purposively selected five active, health-relevant communities&#x2014;r/WomensHealth, r/TwoXChromosomes, r/BirthControl, r/Endo, and r/PCOS (polycystic ovary syndrome)&#x2014;to cover general wellness, social and personal concerns from women&#x2019;s perspectives, contraceptive decision-making, and condition-specific experiences with endometriosis and polycystic ovary syndrome. Focusing on these venues allowed us to observe a broad spectrum of challenges and culturally situated narratives while remaining grounded in communities with sustained engagement and a clear fit to our research aims.</p><p>Data collection used a custom Python pipeline with authenticated API access, retrieving the top-ranked submissions per subreddit across the full subreddit history, along with all nested comments. The specific focus areas and rationales for the 5 selected online communities are detailed in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Full collection parameters are provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. After deduplication and removal of unavailable items, the final corpus comprised 4995 posts and 460,317 comments across the five subreddits. Prior to vectorization, text processing, chunking, and dense embedding configurations were applied as detailed in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>Content safety relied on a multilayer filtering strategy rather than post hoc moderation. At the corpus level, community-curated maximum-salience sampling (top-ranked threads, time_filter=&#x2018;all&#x2019;) inherently prioritized content the community had collectively validated and deprioritized spam and misinformation. Thread chaining via depth-first traversal preserved within-thread corrections, ensuring that conflicting or corrected claims appeared in the same document chunk, as detailed in Section S1.3 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. At the response level, the grounding and attribution instruction in the prompt explicitly prohibited presenting community anecdotes as clinical fact, and a mandatory safety disclaimer was architecturally enforced on every response. When retrieved content conflicted across sources, the prompt instructed the model to surface both perspectives with distinct attribution rather than synthesize them.</p><p>Reddit documents were chunked using a sentence-based strategy with sliding-window overlap to preserve emotional and semantic continuity; full parameters are in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Before chunking, hierarchical comment trees were linearized into single documents via depth-first traversal, as illustrated in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Conversational thread chaining pipeline. This transformation converts Reddit&#x2019;s hierarchical comment trees into linear documents using depth-first traversal, preserving within-thread corrections and emotional context that would be lost by treating comments as isolated posts.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88549_fig02.png"/></fig><p>The second corpus was designed to provide a foundation of factual, expert-vetted medical knowledge to ground the system&#x2019;s responses and prevent the dissemination of misinformation. This corpus was constructed by systematically extracting content from the Mayo Clinic Health System&#x2019;s public-facing website, a trusted source for patient-facing health information.</p><p>An automated web scraping pipeline collected content from key women&#x2019;s health service pages of the Mayo Clinic Health System&#x2019;s public-facing website, including birthing centers, breast cancer care, fertility, mammography, midwifery, obstetrics and gynecology, and prenatal care. The scraper was designed with strict ethical adherence, respecting the site&#x2019;s robots.txt directives and implementing rate limiting to avoid server burden. Full scraping configuration, URL targets, and vector store dimensions are provided in Sections S2.1 and S2.2 and Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-2-3"><title>Retrieval and Response Synthesis</title><p>With the hybrid knowledge base constructed, the next stage of the architecture is the retrieval and fusion engine. This engine is the core of the RAG system, responsible for dynamically identifying and integrating relevant information from both the community and expert corpora in real-time response to a user&#x2019;s query.</p><p>When a user submits a query, it is first passed through the same embedding model used for the knowledge bases. For consistency and to ensure a meaningful comparison in the same vector space, the query is embedded using a model appropriate for the user&#x2019;s likely conversational style. In this implementation, the all-MiniLM-L6-v2 model was used to vectorize the incoming query, projecting it into the same latent space as the Reddit corpus, which most closely resembles natural user language. A key design decision in this architecture is the retrieval of multiple documents rather than a single &#x201C;best&#x201D; match. Traditional RAG systems that retrieve only the top-ranked chunk are often brittle; if that one chunk is noisy, incomplete, or slightly off-topic, the entire generation process can be compromised, leading to a poor or irrelevant response.</p><p>To build robust context, the system retrieves 5 chunks per source (10 total per query) via cosine similarity in FAISS (Facebook AI Similarity Search). This balances GPT-4 context window constraints with informational diversity while mitigating single-chunk brittleness; full configuration is in Section S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Sensitivity analysis across k values remains a priority for future benchmarking.</p><p>The final stage of the architecture is the generation of the chatbot&#x2019;s response. This stage is designed to synthesize the fused context from the retrieval engine into a reply that is helpful, safe, and, crucially, emotionally appropriate for the user&#x2019;s situation. This is achieved through a combination of query analysis and dynamic prompt engineering, leveraging the GPT-4 large language model (LLM) for the final text generation.</p></sec><sec id="s2-2-4"><title>Empathy Mapping and Prompt Engineering</title><p>To move beyond one-size-fits-all replies, the system first infers the user&#x2019;s affective state from their query using the NRC (National Research Council) Emotion Lexicon, computing a length-normalized emotion profile across eight dimensions. NRC configuration details are in Section S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>The empathy goals and their associated response behaviors are defined in <xref ref-type="table" rid="table1">Table 1</xref>. The selected goal configures acknowledgment language, specificity and hedging, and scaffolds (eg, options lists and clinician-question checklists). The tone triad (trust, anticipation, and joy in most contexts) steers lexical choices and framing; retrieval provenance (clinical vs peer narratives) and guardrails operate independently. For auditability, each turn logs the NRC scores, dominant signal, chosen goal, and injected tone.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Empathy goal.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Detected emotion (dominant signals)</td><td align="left" valign="bottom">Empathy goal</td><td align="left" valign="bottom">Response tone (triad)</td><td align="left" valign="bottom">Example</td></tr></thead><tbody><tr><td align="left" valign="top">Distress and frustration&#x2014;fear, sadness, anger</td><td align="left" valign="top">Validate and reassure</td><td align="left" valign="top">Trust, anticipation (with sadness for empathy)</td><td align="left" valign="top">&#x201C;It sounds incredibly frustrating&#x2026; let&#x2019;s look at steps that others found empowering.&#x201D;</td></tr><tr><td align="left" valign="top">Proactive decision-making&#x2014;trust, joy, anticipation, fear</td><td align="left" valign="top">Empower and inform</td><td align="left" valign="top">Trust, joy, anticipation</td><td align="left" valign="top">&#x201C;It&#x2019;s great you&#x2019;re being proactive&#x2026; here&#x2019;s what others and experts recommend.&#x201D;</td></tr><tr><td align="left" valign="top">Ambivalence and vulnerability&#x2014;sadness, disgust, trust</td><td align="left" valign="top">Normalize and support</td><td align="left" valign="top">Trust, anticipation (with sadness)</td><td align="left" valign="top">&#x201C;It&#x2019;s okay to feel conflicted&#x2026; here&#x2019;s factual info to review safely.&#x201D;</td></tr></tbody></table></table-wrap><p>The final step is dynamic prompt construction combining a persona instruction, grounding and attribution rules, an empathy tone directive, retrieved dual-source context, a task instruction, and a mandatory safety disclaimer. The full six-component prompt template is provided in Section S5 and S5.1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>Hallucination mitigation relied on architectural grounding rather than postgeneration filtering: all responses were required to draw on retrieved Mayo Clinic content, and the prompt prohibited unsourced claims. This approach is consistent with prior work by the team and with RAG-based safety strategies documented in the literature. Formal factual consistency evaluation and independent clinical review of outputs were not conducted in this formative study and are recommended as priorities for future validation.</p></sec></sec><sec id="s2-3"><title>Part 2: Feasibility Study Design</title><sec id="s2-3-1"><title>Overview</title><p>We designed the evaluation as an in-the-wild field study to understand how users would naturally interact with the HerCare agent when prompted to explore a women&#x2019;s health topic of personal interest. This approach prioritizes ecological validity, allowing us to observe authentic help-seeking behaviors and capture genuine user reactions to the system&#x2019;s dual-source, empathetic responses.</p><p>We chose a single-condition study design because our primary research questions were descriptive, not comparative. Our goal was to conduct an in-depth characterization of our novel architecture&#x2019;s performance and affective dynamics, rather than to prove its superiority over a baseline. This approach allows for a rich understanding of the user experience with this specific type of system, forming a strong foundation for future comparative work.</p><p>The evaluation assessed three components: feasibility and acceptability via standardized self-report instruments (Chatbot Usability Questionnaire [CUQ], net promoter score [NPS], and postinteraction questionnaire [PIQ]); trustworthiness and empathy via targeted postinteraction ratings; and system affective performance via corpus-level computational linguistic analysis (latent Dirichlet allocation [LDA] topic modeling, VADER [Valence Aware Dictionary and Sentiment Reasoner] sentiment analysis, and NRC emotion profiling) of 1191 conversational turns.</p><p>This study ran for one month (December 18, 2024, to January 17, 2025). The final analytic sample comprised 243 participants who completed all study procedures, which we consider adequate for the quantitative instruments and corpus-level analyses reported here.</p></sec><sec id="s2-3-2"><title>Study Population and Eligibility Criteria</title><p>Participants were required to meet the following inclusion criteria: self-identification as a woman; age between 18 and 45 years; comfort discussing intimate health topics; willingness to have chatbot interactions recorded for research analysis; ability to provide informed digital consent; access to a device with an internet connection; and English-language proficiency. Individuals were excluded if they were younger than 18 years or older than 45 years, did not identify as women, were unwilling to have interactions recorded, lacked internet access, were unable to communicate in English, or had prior involvement in the design or development of the chatbot. These criteria were established to ensure a sample with both the demographic fit and the technological access necessary for a remote, single-session web-based study.</p></sec><sec id="s2-3-3"><title>Recruitment and Sample</title><p>Participants were recruited over the one-month study window via two channels: targeted posts in women&#x2019;s health and wellness communities on social media platforms, and university mailing lists distributed through the research team&#x2019;s professional network. To maximize reach, recruitment posts were distributed across multiple women&#x2019;s health communities on Facebook, Reddit, and Instagram, and reminder posts were issued at two-week intervals throughout the recruitment window.</p><p>Recruitment materials described this study&#x2019;s purpose, the voluntary and confidential nature of participation, and the remote single-session format. No compensation was offered for participation in this phase of this study. Recruitment materials are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s2-3-4"><title>Data Collection Procedure</title><p>This study followed a single-session, fully remote protocol. After arriving at a secure web landing page, participants reviewed study information, provided digital informed consent via checkbox confirmation, and completed a brief demographic questionnaire collecting age range, geographic location, and religious affiliation. They were then directed to the HerCare web interface and given an open-ended naturalistic prompt&#x2014;&#x201C;Please interact with the chatbot about a women&#x2019;s health topic of personal interest. You may ask as many questions as you like&#x201D;&#x2014;designed to avoid priming participants with specific scenarios and to capture authentic help-seeking behavior. Immediately following the interaction, participants completed a fixed-order postinteraction battery: the PIQ, the CUQ, and the NPS. All steps were completed in one sitting, with a median session duration of 16.9 (IQR 11.8-26.3) minutes. The 16.9-minute median duration reflects the combined time for chatbot interaction, PIQ, CUQ, and NPS completion, with no minimum interaction length required. Participants were free to end the chatbot session at any time before proceeding to the surveys. Participants who consented but did not complete the full postinteraction battery were not included in the analytic sample; the participant funnel and a comparison of completers and noncompleters are reported in the Results section.</p></sec><sec id="s2-3-5"><title>Quantitative Survey Instruments</title><p>This section details the three standardized survey instruments used to collect quantitative self-reports from participants. For each instrument, we justify its selection in relation to our research questions and the system&#x2019;s design goals.</p><p>We used standardized, self-report instruments to capture participants&#x2019; perceptions of the system; here we detail the CUQ [<xref ref-type="bibr" rid="ref32">32</xref>]. CUQ is a validated, 16-item questionnaire developed for CAs. Items alternate between positively and negatively worded statements and are rated on a 5-point Likert scale (strongly disagree to strongly agree). Scores are transformed to a 0&#x2010;100 scale, facilitating interpretability and comparison with the widely used System Usability Scale.</p><p>To complement CUQ and NPS with finer-grained diagnostics, we developed a brief PIQ that targets seven facets central to our design goals of trust-calibrated, empathetic assistance: ease of use, helpfulness, clarity, speed, accuracy, trustworthiness, and empathy. The PIQ items and their response scales are presented in <xref ref-type="table" rid="table2">Table 2</xref>. The PIQ was administered immediately after the conversation to capture first-impression judgments at the point of highest salience. Each facet was measured with a single<bold>,</bold> targeted item rated on a 5-point Likert scale, trading breadth for low participant burden and interpretability of construct-specific responses in an in-the-wild setting. The PIQ was developed by the research team specifically for this study to target facets directly relevant to HerCare&#x2019;s design goals; it has not been externally validated, which we acknowledge as a limitation of this formative evaluation.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Postinteraction questionnaire: items and response scales.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Questions presented to participants</td><td align="left" valign="bottom">5-point Likert scale anchors</td></tr></thead><tbody><tr><td align="left" valign="top">Ease of use</td><td align="left" valign="top">&#x201C;How easy or difficult was it to interact with the chatbot?&#x201D;</td><td align="left" valign="top">1=extremely difficult, 5=extremely easy</td></tr><tr><td align="left" valign="top">Helpfulness</td><td align="left" valign="top">&#x201C;How helpful or unhelpful was the information provided by the chatbot?&#x201D;</td><td align="left" valign="top">1=extremely unhelpful, 5=extremely helpful</td></tr><tr><td align="left" valign="top">Clarity</td><td align="left" valign="top">&#x201C;How clear or unclear were the chatbot&#x2019;s responses?&#x201D;</td><td align="left" valign="top">1=extremely unclear, 5=extremely clear</td></tr><tr><td align="left" valign="top">Speed</td><td align="left" valign="top">&#x201C;How would you rate the speed of the chatbot&#x2019;s responses?&#x201D;</td><td align="left" valign="top">1=extremely slow, 5=extremely fast</td></tr><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">&#x201C;How accurate or inaccurate did you perceive the information from the chatbot to be?&#x201D;</td><td align="left" valign="top">1=extremely inaccurate, 5=extremely accurate</td></tr><tr><td align="left" valign="top">Trustworthiness</td><td align="left" valign="top">&#x201C;How trustworthy or untrustworthy did you find the chatbot?&#x201D;</td><td align="left" valign="top">1=extremely untrustworthy, 5=extremely trustworthy</td></tr><tr><td align="left" valign="top">Empathy</td><td align="left" valign="top">&#x201C;How empathetic or unempathetic did the chatbot seem in its responses?&#x201D;</td><td align="left" valign="top">1=extremely unempathetic, 5=extremely empathetic</td></tr></tbody></table></table-wrap><p>Regarding scoring, for each metric, we summarize responses from all completers (N=243), with higher values indicating more positive assessments; item wording and anchors enable straightforward replication. Distributional normality was assessed for the CUQ total and each PIQ facet using the Shapiro-Wilk test. All measures departed significantly from normality (all <italic>P</italic>&#x003C;.001) and were left-skewed, reflecting ceiling effects typical of high-satisfaction Likert data. We therefore report medians with IQRs as the primary descriptive statistics, retaining means and SDs only to aid comparison with prior literature.</p></sec><sec id="s2-3-6"><title>Computational Analysis of Conversational Data</title><p>This section transitions from self-reported survey data to the objective, computational analysis of the raw conversational logs generated during this study. A total of 1191 user-agent conversational turns were collected from the 243 participants and subjected to analysis. This approach was chosen to complement and triangulate the subjective survey findings with objective, scalable linguistic analysis. Three specific natural language processing techniques were used to explore the nature of the interactions.</p><p>Survey responses were analyzed descriptively. As reported above, distributional normality was assessed with the Shapiro-Wilk test; because all CUQ and PIQ measures were nonnormal and left-skewed, we report medians with IQRs as the primary statistic, with means and SDs for comparison. NPS was computed as the percentage of promoters minus detractors. For the conversational corpus, the three natural language processing techniques were applied as follows: LDA first partitioned the corpus into latent topics, and VADER sentiment scores and NRC emotion profiles were then computed for each turn and aggregated by LDA-derived topic, enabling the stratified, per-topic comparisons reported in the Results section. All analyses were conducted in Python.</p><p>We used LDA to discover themes in the conversational corpus without imposing predefined categories. LDA is a generative probabilistic model for topic modeling that assumes each document&#x2014;in our case, a participant&#x2019;s full conversation with the agent&#x2014;is a mixture of latent topics, and each topic is a probability distribution over words. Operating under a bag-of-words assumption, LDA learns both (1) topics represented by characteristic keywords and (2) per-document topic proportions, enabling us to summarize what was discussed and how strongly each theme appeared within a conversation.</p><p>We used VADER to measure sentiment polarity in both user queries and system replies. VADER is a lexicon- and rule-based tool tuned for informal text (eg, social media and conversational language). Its human-validated lexicon assigns valence scores (&#x2212;4 to +4) to words, emoticons, and slang, while a rule engine adjusts for pragmatic cues such as intensifiers (&#x201C;very&#x201D; and &#x201C;extremely&#x201D;), punctuation and capitalization for emphasis, and negation (&#x201C;not happy&#x201D;). For any input, VADER returns proportions of positive, neutral, and negative sentiment and a normalized compound score in [&#x2212;1, 1].</p><p>We used the NRC Emotion Lexicon (EmoLex) to characterize the emotional composition of dialogues at a granularity beyond simple polarity. EmoLex comprises 14k+ English terms annotated&#x2014;via large-scale crowdsourcing by the NRC of Canada&#x2014;with associations to eight basic emotions (anger, fear, anticipation, trust, surprise, sadness, joy, and disgust) and two sentiments (positive and negative). For each text span (eg, a user turn or system reply), we tokenize, look up lexicon matches, and aggregate counts per emotion; counts are length-normalized to yield a multidimensional emotion profile that captures the affective fingerprint of the utterance.</p></sec></sec><sec id="s2-4"><title>Ethical Considerations</title><p>This study&#x2019;s protocol was reviewed and approved by the Institutional Review Board of North Dakota State University (Protocol #IRB0005368, approved December 11, 2024, expiration December 10, 2027; Exempt, Category 3&#x2014;Benign Behavioral Interventions). A protocol amendment covering additions to a subsequent phase of the broader study was approved on August 12, 2025. The research is supported by the NSF (National Science Foundation) RII (research infrastructure improvement) Track 2 FEC (focus areas competitive) grant. Participation was entirely voluntary. All prospective participants reviewed a digital information sheet detailing study procedures, potential risks, and benefits, and provided digital informed consent via checkbox confirmation on a secure online platform before any data collection. All study data were stored on encrypted, password-protected institutional servers and analyzed using deidentified datasets in which unique participant IDs replaced all personal identifiers. Research records will be retained for three years following study completion in accordance with North Dakota State University Institutional Review Board requirements, after which personally identifiable information will be permanently deleted.</p><p>The construction of the peer-support knowledge base involved data collected from public Reddit communities. Data were collected exclusively via the official Reddit API using the PRAW (Python Reddit API Wrapper), in compliance with Reddit&#x2019;s terms of service. We adhered to safe harbor ethical guidelines for internet research and treated all content as sensitive given the intimate health nature of the topics discussed. A strict deidentification pipeline stripped author usernames, user flair, and specific timestamps during data ingestion. The system synthesizes retrieved narratives as aggregate, anonymous community perspectives, preventing reidentification of individual posters in the chatbot&#x2019;s output. No attempt was made to contact original posters or interact with the platform beyond read-only data access.</p><p>Given the risks inherent in AI-mediated health information, the system was engineered with layered safety guardrails. Architecturally, retrieval from the Mayo Clinic corpus ensured that every response was grounded in vetted clinical content. At the prompt level, the model was explicitly prohibited from presenting community advice as medical fact and was required to attribute all information to its source. As a mandatory fail-safe, every response concluded with a safety disclaimer explicitly stating that HerCare is not a doctor and that the information provided is for educational support only and is not a substitute for professional medical advice.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>This section moves from user acceptance (CUQ, NPS, and PIQ) to corpus-level dialogue analyses, quantifying sentiment polarity and decomposing emotion profiles (VADER and NRC) by comparing user inputs with system replies. We conclude by synthesizing these layers to describe how a trust-calibrated, empathy-oriented design manifests in practice.</p></sec><sec id="s3-2"><title>Overall System Performance and User Acceptance</title><p>The multistage pipeline and architectural integration of peer and clinical knowledge sources evaluated during this field study are illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>. Regarding participant funnel, the public, uncompensated study link was accessed by 415 visitors during the one-month window (December 18, 2024, to January 17, 2025). The integration of this evaluation phase within the overall HerCare design and development workflow is depicted in <xref ref-type="fig" rid="figure3">Figure 3</xref>. <xref ref-type="fig" rid="figure3">Figure 3</xref> provides an overview of the five-phase design and development process for HerCare. Phases 2a and 2b were conducted in parallel, reflecting an asymmetric corpus construction strategy in which each knowledge source received a distinct processing pipeline and embedding model optimized for its data type. Not all were participants in the intended sense: 80 did not provide eligible, informed consent (including 4 who were ineligible by age), consistent with casual exploration of a publicly advertised chatbot. Of the remaining 335 eligible, consenting visitors, 243 (72.5%) completed the full protocol (chatbot interaction, PIQ, CUQ, and NPS) and constitute the analytic sample; 92 (27.5%) did not complete the survey battery.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>HerCare design and development process. CUQ: chatbot usability questionnaire; FAISS: Facebook AI similarity search; LDA: latent Dirichlet allocation; NPS: net promoter score; NRC: National Research Council; PIQ: postinteraction questionnaire; RAG: retrieval-augmented generation; VADER: Valence Aware Dictionary and Sentiment Reasoner.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88549_fig03.png"/></fig><p>Regarding participant cohort, analyses draw on 243 completers (from 335 eligible, consenting visitors) in a single-session, remote study (median duration &#x2248;16.9, IQR 11.8-26.3 min).</p><p>The sample skewed younger, predominantly US-based and Christian, with smaller Muslim and other representation (<xref ref-type="table" rid="table3">Table 3</xref>). This context informs the interpretation of acceptance and satisfaction scores.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Participant demographics and overall usability metrics with comparison of completers and consented noncompleters.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Participant demographics (N=243)</td><td align="left" valign="bottom">Completers (n=243)</td><td align="left" valign="bottom">Noncompleters (n=92)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (years), n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">.11</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>18&#x2010;27</td><td align="left" valign="top">117 (48.1)</td><td align="left" valign="top">53 (57.6)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>28&#x2010;37</td><td align="left" valign="top">107 (44.0)</td><td align="left" valign="top">29 (31.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>38&#x2010;45</td><td align="left" valign="top">19 (7.8)</td><td align="left" valign="top">10 (10.9)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Religion, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">.52</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Christianity</td><td align="left" valign="top">195 (80.2)</td><td align="left" valign="top">70 (76.1)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Islam</td><td align="left" valign="top">39 (16.0)</td><td align="left" valign="top">18 (19.6)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hinduism</td><td align="left" valign="top">2 (0.8)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Judaism</td><td align="left" valign="top">2 (0.8)</td><td align="left" valign="top">1 (1.1)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Other/multiple</td><td align="left" valign="top">5 (2.1)</td><td align="left" valign="top">3 (3.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Geography</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">.052</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>United States</td><td align="left" valign="top">203 (83.5)</td><td align="left" valign="top">68 (73.9)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other</td><td align="left" valign="top">40 (16.5)</td><td align="left" valign="top">24 (26.1)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Session duration</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Median (IQR), minutes</td><td align="left" valign="top">16.9 (11.8&#x2010;26.3)</td><td align="left" valign="top">6.1 (3.6&#x2010;9.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">System evaluation metrics</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CUQ<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>: mean (SD)</td><td align="left" valign="top">75.67 (15.50)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CUQ: median (IQR)</td><td align="left" valign="top">78.1 (65.2&#x2010;87.5)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CUQ: range</td><td align="left" valign="top">37.5&#x2010;100</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Net promoter score</td><td align="left" valign="top">60.0</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Not applicable.</p></fn><fn id="table3fn2"><p><sup>b</sup>CUQ: Chatbot Usability Questionnaire.</p></fn></table-wrap-foot></table-wrap><p>To check for selection bias from the open recruitment, we compared completers (n=243) with consented noncompleters (n=92) on all prediscontinuation characteristics (<xref ref-type="table" rid="table3">Table 3</xref>). The groups did not differ in age (<italic>P</italic>=.11), religion (<italic>P</italic>=.52), or country (<italic>P</italic>=.052), but completers had longer sessions (median 16.9, IQR 11.8-26.3 vs 6.1, IQR 3.6-9.5 min; <italic>P</italic>&#x003C;.001) and noncompleters exited at a median 40% survey progress, almost all at the questionnaire stage rather than during the chat. No noncompleter produced a complete CUQ and only one provided a PIQ or NPS response, so no analyzable partial data could be included. The shorter sessions and uniform survey-stage exit suggest many noncompleters sampled the interface out of curiosity rather than engaging in good-faith help-seeking; comparable demographics bound, without excluding, self-selection (see Limitations section).</p><p>The agent&#x2019;s CUQ median was 78.1 (IQR 65.2&#x2010;87.5; mean 75.67, SD 15.50), indicating that most participants experienced the system as good to excellent in usability. The distribution was left-skewed (Shapiro-Wilk <italic>P</italic>&#x003C;.001), with ratings concentrated at the upper end of the scale.</p><p>The seven PIQ facets clarify why top-line scores are strong. <xref ref-type="table" rid="table4">Table 4</xref> gives details of the scores of the PIQ. The highest ratings&#x2014;helpfulness<bold>,</bold> ease of use<bold>,</bold> and clarity&#x2014;indicate that participants could quickly orient, obtain understandable guidance, and feel the interaction was productive. Across all facets, the median response was at the top or near-top of the scale (<xref ref-type="table" rid="table4">Table 4</xref>); accuracy, trustworthiness, and empathy showed the most spread (median 4, IQR 4&#x2010;5), suggesting these relational and correctness dimensions are judged somewhat more stringently than ease and helpfulness.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Postinteraction satisfaction scores (N=243). Facets were left-skewed (Shapiro-Wilk <italic>P</italic>&#x003C;.001); medians with IQR are the primary statistic, with means and SDs reported for comparison. Scale: 1&#x2010;5, higher is more positive.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Satisfaction metric</td><td align="left" valign="bottom">Median (IQR)</td><td align="left" valign="bottom">Average (1-5; SD)</td></tr></thead><tbody><tr><td align="left" valign="top">Helpfulness</td><td align="left" valign="top">5 (4-5)</td><td align="left" valign="top">4.55 (0.65)</td></tr><tr><td align="left" valign="top">Ease of use</td><td align="left" valign="top">5 (4-5)</td><td align="left" valign="top">4.49 (0.80)</td></tr><tr><td align="left" valign="top">Clarity</td><td align="left" valign="top">5 (4-5)</td><td align="left" valign="top">4.47 (0.66)</td></tr><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">4 (4-5)</td><td align="left" valign="top">4.39 (0.63)</td></tr><tr><td align="left" valign="top">Speed</td><td align="left" valign="top">5 (4-5)</td><td align="left" valign="top">4.33 (0.76)</td></tr><tr><td align="left" valign="top">Trustworthiness</td><td align="left" valign="top">4 (4-5)</td><td align="left" valign="top">4.35 (0.60)</td></tr><tr><td align="left" valign="top">Empathy</td><td align="left" valign="top">4 (4-5)</td><td align="left" valign="top">4.17 (0.74)</td></tr></tbody></table></table-wrap></sec><sec id="s3-3"><title>The Conversation Polarity Shift (VADER)</title><p>A VADER analysis of 1191 query-response pairs shows a clear, topic-general polarity shift from user inputs to assistant replies, illustrated in <xref ref-type="fig" rid="figure4">Figures 4</xref><xref ref-type="fig" rid="figure5"/>-<xref ref-type="fig" rid="figure6">6</xref> below.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>The consistent gap between neutral or negative user sentiment and strongly positive assistant sentiment is evident across all categories. Consistent with the system&#x2019;s design, agent responses to distress-laden user language were uniformly supportive and forward-looking regardless of topic. PMS: premenstrual syndrome; VADER: Valence Aware Dictionary and Sentiment Reasoner.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88549_fig04.png"/></fig><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>User query sentiment proportions and compound polarity scores by topic (VADER analysis). Neutral sentiment dominates across all topics (&#x003E;0.74), concentrated in the informational and symptom-oriented LDA topics. Compound scores range from +0.154 (birth control options and decision-making) to &#x2212;0.181 (stress, anxiety, and coping strategies), with distress-laden topics registering the most negative user language. Full per-topic data are provided in Table S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. LDA: latent Dirichlet allocation; Mgmt: management; PMS: premenstrual syndrome; VADER: Valence Aware Dictionary and Sentiment Reasoner.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88549_fig05.png"/></fig><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Assistant response sentiment proportions and compound polarity scores by topic (VADER analysis). In direct contrast to <xref ref-type="fig" rid="figure5">Figure 5</xref>, all compound scores are strongly positive (+0.555 to +0.826), and negative sentiment proportions are uniformly low (0.018&#x2010;0.069). Topics with the most negative user queries&#x2014;stress, anxiety, and coping strategies, and menstrual cramps and pain experiences&#x2014;still elicit robustly positive assistant responses, demonstrating that the polarity shift holds under the most affectively challenging conditions. Full per-topic data are provided in Table S2 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. PMS: premenstrual syndrome.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88549_fig06.png"/></fig><p>Overall, user turns skew neutral-to-negative, which is expected in a help-seeking setting. Across topics, the neutral component dominates (often &#x003E;0.80). This neutral dominance is concentrated in the LDA-derived topics whose keyword distributions are informational and symptom-oriented (eg, menstrual cycle concerns, menstrual cramps and pain experiences, and the information-seeking facet of birth control options and decision-making), where users predominantly describe symptoms and request factual guidance; we therefore read the neutral component as fact-seeking and symptom description rather than affective flatness. Negative sentiment, by contrast, rises in the distress-laden topics. The lowest compound score appears for stress, anxiety, and coping strategies (compound &#x2212;0.181), followed by menstrual cramps and pain experiences (compound &#x2212;0.131), indicating language of concern, discomfort, and uncertainty. In contrast, more proactive themes register higher positivity; birth control options and decision-making show the most positive user compound (+0.154), consistent with forward-looking, choice-oriented discourse.</p><p>In marked contrast, assistant turns are consistently and strongly positive across every topic. Compound scores range from +0.555 to +0.826, with positive proportions markedly higher than in user inputs (<xref ref-type="fig" rid="figure5">Figure 5</xref>). Negative sentiment in replies is low across the board (0.018&#x2010;0.069), even when addressing sensitive topics such as menstrual pain or stress. This pattern indicates that responses systematically emphasize supportive, reassuring language while avoiding alarming or discouraging phrasing. Notably, topics that begin with the most negative user tone (eg, stress, anxiety, and coping strategies, and menstrual cramps and pain experiences) still elicit robustly positive assistant compounds (&#x2265;+0.554), confirming that the system maintained a positive, supportive stance even under the most affectively challenging conditions.</p><p>This systematic gap between user-query sentiment (concern) and agent-response sentiment (support) is evident topic by topic across <xref ref-type="fig" rid="figure4">Figures 4</xref><xref ref-type="fig" rid="figure5"/>-<xref ref-type="fig" rid="figure6">6</xref>, which contrast the aggregate sentiment scores of user queries and assistant responses for each topic.</p><p>Taken together, the VADER results depict a stable user&#x2192;agent positivity transition: users frequently open with neutral descriptions punctuated by negative affect in distress topics, and the agent replies with uniformly positive, supportive language across all themes. The contrast is visible topic-by-topic (<xref ref-type="fig" rid="figure4">Figures 4</xref><xref ref-type="fig" rid="figure5"/>-<xref ref-type="fig" rid="figure6">6</xref>): the lowest user compounds remain negative while the corresponding assistant compounds fall well into the positive range, and proactive topics show positivity on both sides with a further uplift in replies. We refrain from causal claims, and we note that these analyses pool user and agent turns rather than tracking individual sessions over time; the results therefore describe an aggregate difference between user-query sentiment and agent-response sentiment, not a within-session trajectory.</p></sec><sec id="s3-4"><title>Decoding the Emotion Dialogue</title><p>Using the NRC Emotion Lexicon, we profiled eight emotions&#x2014;anger, anticipation, disgust, fear, joy, sadness, surprise, trust&#x2014;for each turn and summarized them by topic (<xref ref-type="fig" rid="figure7">Figures 7</xref> and <xref ref-type="fig" rid="figure8">8</xref>). This analysis moves beyond polarity to describe the affective texture of the dialogue, indicating what users bring into the conversation and how the agent responds.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>NRC emotion profiles of user queries by topic. Warmer colors indicate higher normalized emotion scores. Three distinct affective profiles are visible: a proactive profile (birth control and decision-making, top row) with elevated trust (0.966), joy (0.940), and anticipation (0.906) alongside high fear (0.846); a distress profile (stress, anxiety, and coping strategies) with elevated anger (0.356), fear (0.444), and sadness (0.437); and an ambivalence profile (sexual and intimate health in religious contexts) with elevated disgust (0.203) and sadness (0.351) alongside moderate trust (0.500). Full per-topic scores are provided in Table S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. These three distinct profiles confirm that users brought meaningfully different emotional needs to their interactions with HerCare, providing empirical grounding for the system&#x2019;s empathy-mapping design. NRC: National Research Council; PMS: premenstrual syndrome.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88549_fig07.png"/></fig><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>NRC emotion profiles of assistant responses by topic. The cooler blue scale reflects the higher absolute magnitude of scores relative to user queries (<xref ref-type="fig" rid="figure7">Figure 7</xref>). Trust dominates every row (range: 3.523&#x2010;5.281), peaking at 5.281 for stress, anxiety, and coping strategies. The validate-then-redirect pattern is visible in the menstrual cramps and pain experiences row, where sadness (1.402) is elevated alongside high trust (3.825) and anticipation (1.567), indicating that the agent briefly mirrors user distress before pivoting to constructive guidance. Full per-topic scores are provided in Table S4 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. The uniform trust dominance across all topics, combined with targeted sadness elevation in distress contexts, confirms that the empathy-mapping layer successfully operationalized the validate-then-redirect strategy consistently at scale. NRC: National Research Council; PMS: premenstrual syndrome.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e88549_fig08.png"/></fig><p>Regarding user emotional landscape, overall, user language reflects three recurring configurations. First, a proactive profile appears in birth control options and decision-making, where positive emotions co-occur at elevated levels&#x2014;trust 0.966, joy 0.940, and anticipation 0.906&#x2014;alongside substantial fear 0.846, consistent with high-stakes, forward-looking decision-making. Second, a distress profile characterizes stress, anxiety, and coping strategies&#x2014;fear 0.444, sadness 0.437, and anger 0.356&#x2014;and pain-centered topics such as menstrual cramps and menstrual cycle concerns, where fear and sadness are high, and anger is comparatively lower, indicating discomfort and uncertainty rather than frustration. Third, an ambivalence profile emerges in sexual and intimate health in religious contexts, which combines elevated sadness 0.351 and disgust 0.203 with moderate trust 0.500, suggesting emotionally conflicted help-seeking. Across topics, these distributions provide a concise map of users&#x2019; affective needs: hopeful yet apprehensive in choice-making; distressed in pain and anxiety; and conflicted when intimate concerns intersect with values and norms.</p><p>Regarding assistant emotional profile, in contrast, the agent&#x2019;s replies are dominated by trust across nearly all topics, with consistently high anticipation and joy&#x2014;a positivity triad that recurs throughout the corpus (<xref ref-type="fig" rid="figure8">Figure 8</xref>). Trust often exceeds other emotions by a wide margin, peaking at 5.281 for stress, anxiety, and coping strategies and 4.902 for period-related mood swings and symptom management. Negative emotions are present but comparatively low; importantly, sadness appears in contexts where acknowledgment is appropriate. For instance, in menstrual cramps and pain experiences, the agent shows its highest sadness 1.402, yet, this acknowledgment co-occurs with higher trust 3.825 and anticipation 1.567. This pattern is consistent with a validate-then-redirect response: the reply briefly reflects user distress (validation) before shifting emphasis toward confidence and forward-looking guidance (redirection). We observe this configuration across topics, including those with the most distressed user profiles, indicating a stable, supportive affective stance in the replies.</p><p>Taken together, the NRC results provide a granular description of how the emotional composition of conversations differs between users and the agent: users present with blends of fear, sadness, and (in places) anger or ambivalence, while replies emphasize trust and anticipation, with targeted acknowledgment where distress is salient. <xref ref-type="fig" rid="figure7">Figures 7</xref> and <xref ref-type="fig" rid="figure8">8</xref> above visualize these per-topic patterns; full numerical data are provided in Tables S3 and S4 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This formative study developed and evaluated HerCare, a dual-source RAG CA for women&#x2019;s health. Across standardized usability instruments, postinteraction ratings, and corpus-level computational linguistic analysis, participants who completed this study rated the system highly on usability, trustworthiness, helpfulness, and empathy, consistent with good acceptability in this completer sample. Computational analysis of 1191 conversational turns confirmed that the system&#x2019;s linguistic behavior aligned with its design goals: user queries skewing neutral-to-negative in affect were consistently met with strongly positive, trust-dominant responses. The designed validate-then-redirect behavior was consistently confirmed across topics, with the agent briefly acknowledging user distress before pivoting toward constructive guidance&#x2014;an effective strategy enacted consistently without explicit per-topic tuning.</p></sec><sec id="s4-2"><title>Quantifying Empathy Insights From Computational Affective Analysis</title><p>The high self-reported empathy score of 4.17 of 5, together with corpus-level evidence of a systematic polarity gap and trust-dominant agent responses, is consistent with HerCare enacting its empathic design goals at the linguistic level. Our work contributes a methodological approach for looking beyond self-report scores to characterize how empathy is expressed. By triangulating user perceptions with a computational analysis of the dialogue itself, we obtain a more granular, though still descriptive, view of the system&#x2019;s affective behavior.</p><p>The VADER analysis provided a clear, quantitative measure of the polarity gap between user queries and agent responses. That the agent replied to neutral or negative queries with strongly positive language confirms the system behaved as designed; the contribution here is methodological rather than an independent behavioral discovery, in that it provides a blueprint for how researchers can quantify an agent&#x2019;s response sentiment relative to user input.</p><p>The NRC emotion analysis allowed for an even deeper insight, revealing the specific texture of the agent&#x2019;s persona. We observed that the AI was not generically happy but appeared to act as a trust engine. Its emotional output was overwhelmingly dominated by the language of trust and anticipation, a strategic choice designed to instill confidence and hope. This observation, made possible only through computational analysis, suggests that the system&#x2019;s empathetic goals were successfully translated into a measurable linguistic strategy. The most informative cross-method observation is the convergence between the agent&#x2019;s elevated NRC trust and anticipation scores and participants&#x2019; independently reported empathy ratings: two methods drawing on different data sources&#x2014;one a corpus-level lexical count of the agent&#x2019;s output, the other a subjective postinteraction judgment by users&#x2014;point in the same direction. We frame this carefully. The convergence does neither establish that the NRC profile caused the empathy ratings, nor that NRC trust is a validated proxy for perceived empathy; the two were not statistically correlated at the participant level, since NRC scores are corpus-aggregated rather than per-participant. What it does offer is triangulation: a self-report signal and a behavioral-linguistic signal that would not necessarily agree but nonetheless align, lending convergent (though not confirmatory) support to the claim that the agent&#x2019;s designed affective strategy was perceptible to users. We regard establishing a per-participant link between specific emotion profiles and perceived empathy as a valuable direction for future, appropriately powered work. This multimethod quantitative approach&#x2014;combining subjective user ratings with corpus-level linguistic analysis&#x2014;offers a practical framework for future digital health research aiming to evaluate the affective qualities of empathetic systems.</p></sec><sec id="s4-3"><title>Validate, Then Redirect: A Design Pattern for Supportive AI</title><p>A descriptive contribution of our work is the characterization of an empathy strategy that the agent was designed to enact. The question is not just that users found the agent empathetic, but how that empathy was expressed at the linguistic level. The NRC analysis makes this concrete, confirming a consistent pattern we term validate, then redirect.</p><p>This pattern is most visible in distressing topics. For instance, in menstrual-pain discussions, the agent&#x2019;s responses registered a notable level of sadness, consistent with acknowledging the user&#x2019;s negative experience, alongside a substantially larger measure of trust and anticipation that shifts the response toward a constructive, hopeful frame. Rather than simple emotional mirroring, this resembles a multistep supportive-communication strategy: it signals &#x201C;I hear your pain&#x201D; before orienting the user toward potential solutions.</p><p>We propose validate, then redirect as a concrete and transferable design pattern for supportive AI. That this strategy was consistently operationalized by our architecture&#x2014;and that users responded positively&#x2014;suggests that with the right architectural foundation and prompting strategy, LLM-based systems can move beyond generic positivity toward context-aware emotional support. The articulation and refinement of such reusable design patterns, rather than their discovery as novel behaviors, is what we offer as a contribution toward more human-centered health AI.</p><p>Beyond the evaluation findings, the development of HerCare surfaced several design challenges and considerations that are broadly transferable to future empathy-attuned health AI systems. The most fundamental challenge was reconciling two epistemically different knowledge sources, peer narratives and clinical guidance, within a single response without falsely synthesizing them into a unified authoritative voice. The solution was architectural transparency: rather than merging sources into a seamless reply, the system was engineered to surface provenance explicitly in every response, allowing users to evaluate community and clinical content on their own terms. A second challenge was engineering affective responsiveness without requiring labeled emotional training data or fine-tuning: in our implementation, the NRC-based empathy-mapping layer&#x2014;lexical inference over a standardized emotion lexicon combined with prompt-level goal injection&#x2014;was associated with measurable empathic linguistic behavior across the corpus, suggesting this lighter-weight approach is a feasible alternative to supervised emotional modeling. Third, content quality control for Reddit data required a shift from post hoc moderation to corpus-level design relying on community voting, longitudinal engagement filters, and thread chaining to surface high-signal, collectively validated narratives rather than applying blanket content rules. Together, these design choices suggest a replicable pattern for building empathy-oriented health AI: treat knowledge provenance as a trust mechanism, use inference-based affective adaptation rather than supervised learning, and leverage community curation as a quality proxy.</p></sec><sec id="s4-4"><title>Comparison With Prior Work</title><p>Early reviews of CAs in health care showed promise but highlighted limited evidence for behavior change, small or quasi-experimental evaluations, and scarce safety reporting [<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref35">35</xref>]. More recent syntheses continue to stress gaps in safety, evaluation quality, and appropriateness for sensitive contexts such as reproductive and intimate health [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. Parallel work in women&#x2019;s digital health (femtech) documents generic guidance, gaps in menopause support, and serious post-Roe privacy risks in period or fertility apps [<xref ref-type="bibr" rid="ref38">38</xref>-<xref ref-type="bibr" rid="ref44">44</xref>]. Studies of menstrual health apps further critique narrow personas, heteronormative defaults, and insufficient clinical grounding, despite high uptake and perceived utility [<xref ref-type="bibr" rid="ref45">45</xref>-<xref ref-type="bibr" rid="ref49">49</xref>]. A recent decade review consolidates these concerns, documenting Western-centric defaults and persistent underserving of Global South users, migrants, and marginalized genders, and advances a Reproductive Well-Being for All framework as a corrective lens for future systems [<xref ref-type="bibr" rid="ref50">50</xref>]. Together, this work motivates systems that deliver accurate guidance while addressing the emotional and contextual realities of women&#x2019;s health. These review gaps suggest that how information is delivered&#x2014;its empathy and cultural fit&#x2014;may be as consequential as correctness, motivating work on relational, tailored interactions.</p><p>Our finding that empathy ratings (4.17/5) were slightly lower than functional ratings aligns with this literature&#x2019;s observation that relational qualities are judged more stringently than task performance, reinforcing the importance of investing in empathic design beyond surface-level pleasantries. Relational and empathic interaction has long been associated with better engagement, alliance, satisfaction, and even clinical outcomes [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. HCI work on relational agents demonstrated that explicitly designed empathic behaviors (eg, responsiveness to emotion and social dialogue) can build durable alliances in health contexts [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. Meta-analytic and theoretical work in health communication shows that personalization/tailoring reliably improves outcomes relative to generic messaging&#x2014;when tailored to user characteristics and needs [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]. Public-health scholarship further distinguishes surface vs deep cultural tailoring [<xref ref-type="bibr" rid="ref7">7</xref>]: beyond language and imagery, effective interventions embed community values, norms, and lived realities [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. Within women&#x2019;s intimate health, HCI studies foreground religious and cultural values shaping help-seeking and design needs, including privacy/modesty considerations and taboo navigation [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]. As a concrete Global South system example, HCI researchers co-designed a culturally appropriate AI combining community participation, professional moderation, and language sensitivity to improve fit while surfacing governance trade-offs designers must anticipate [<xref ref-type="bibr" rid="ref57">57</xref>]. These strands collectively argue that conversational health systems should adapt to a user&#x2019;s emotional state, background, and cultural context&#x2014;not only to what is asked but to how it is asked. Yet, even carefully tailored clinical guidance can miss lived, context-specific concerns; online peer communities surface this missing layer.</p><p>HerCare&#x2019;s dual-source architecture operationalizes what the peer support literature has long documented: that situated community knowledge complements clinical sources in ways that increase perceived relevance, particularly around taboo or culturally sensitive concerns. A large body of HCI or computer-supported cooperative work literature shows that peer spaces (eg, Reddit, WhatsApp [Meta], and Facebook groups) provide social support, reciprocity, and stigma-safe disclosure for sensitive topics including sexual abuse, postpartum depression, and broader reproductive health [<xref ref-type="bibr" rid="ref58">58</xref>-<xref ref-type="bibr" rid="ref64">64</xref>]. These communities often supply practical, situated knowledge that complements clinical guidance and increases perceived relevance, particularly around taboo or culturally sensitive concerns [<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref63">63</xref>]. The design of digital safe spaces&#x2014;including moderation, anonymity, and culturally appropriate norms&#x2014;has been shown to be critical for women and gender minorities navigating health taboos [<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref61">61</xref>]. This motivates architectures that can responsibly retrieve peer-support narratives alongside vetted clinical sources, while maintaining transparency about provenance and scope.</p><p>Users reported high trust in the system (4.35/5, 93% rating it trustworthy), a system that used explicit source attribution throughout. We did not, however, isolate the effect of attribution itself: participants were not asked whether attribution influenced their trust, so we cannot attribute the high trust ratings to attribution specifically. This remains an open question, of particular interest given the mixed findings on AI labeling effects in this literature. LLMs can exhibit strong performance on medical benchmarks; yet, reliability varies by task and evaluation practice&#x2014;and human evaluations in health care remain uneven in design rigor [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref66">66</xref>]. RAG improves factuality and auditability by grounding responses in citable retrieved sources; yet, AI labeling has nuanced effects&#x2014;labels can dampen perceived empathy and accuracy in some settings and often have limited impact on persuasiveness [<xref ref-type="bibr" rid="ref67">67</xref>-<xref ref-type="bibr" rid="ref69">69</xref>]. At the same time, disclosure and labeling present nuanced effects: AI-generated responses can make people feel heard, but explicit AI labeling can dampen perceived empathy or quality in some settings [<xref ref-type="bibr" rid="ref70">70</xref>]; other studies find labeling has minimal impact on persuasiveness depending on context [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref70">70</xref>-<xref ref-type="bibr" rid="ref72">72</xref>]. Collectively, the literature suggests health chatbots should (1) ground responses in transparent, citable sources (clinical + community where appropriate), (2) communicate uncertainty, and (3) calibrate trust through honest signaling of capabilities or limitations.</p></sec><sec id="s4-5"><title>Limitations</title><p>While our study offers preliminary evidence for the feasibility of the HerCare architecture, we acknowledge several limitations inherent to its formative nature. Our goal in this work was to characterize a novel system and establish feasibility; the following points outline the necessary next steps to build upon this foundation.</p><p>Most importantly, all outcomes are based on completers, an inherent limitation of this uncompensated, single-session design: 92 of 335 (27.5%) consenting visitors did not finish the survey battery. Completers and noncompleters did not differ significantly in measured demographics, and noncompleters overwhelmingly exited at the survey stage rather than during the interaction, but self-selection toward more engaged or favorably disposed users cannot be excluded. Usability, trustworthiness, empathy, and NPS estimates should thus be read as completer-only and may be optimistic; the NPS of 60, in particular, could be lower across all visitors. No analyzable partial data were available for noncompleters. Future deployments should add lightweight interaction measures and modest incentives to reduce survey-stage attrition.</p><p>First, this formative study provides evidence of the system&#x2019;s positive reception but cannot make causal claims about its effectiveness relative to other approaches. The single-condition field study design, while maximizing ecological validity, does not include a control group. Future work should conduct a randomized controlled trial comparing our dual-source architecture against several key baselines: a single-source system using only expert clinical data, a single-source system using only peer narratives, and a generic, nonretrieval-based LLM with a similar empathetic persona. This would allow us to isolate the impact of the dual-source RAG architecture on user trust and perceived empathy.</p><p>Second, the sample&#x2019;s demographic composition of the predominantly young (18-37 years), US-based (203/243, 83.5%), and Christian (195/243, 80.2%) limits generalizability in a domain where culture and religion directly shape help-seeking behavior. Women&#x2019;s health experiences vary substantially across religious and cultural contexts, particularly around contraception, reproductive choices, and sexual health. The extent to which HerCare&#x2019;s empathic strategies translate to non-Western, non-Christian, or older populations remains an open empirical question that future work must address through purposive sampling of underrepresented groups.</p><p>Third, both VADER and the NRC Emotion Lexicon are lexicon-based tools with known constraints in contextual sensitivity, cultural variability, and complex negation handling. They cannot detect sarcasm, irony, or culturally situated emotional expression, and their lexicons reflect primarily Western, English-language cultural associations. Findings from these analyses should be interpreted as descriptive corpus-level patterns rather than clinically precise affective measurements.</p><p>Fourth, the PIQ was developed by the research team specifically for this study and has not been externally validated. While its facets converged with validated instruments (CUQ and NPS), the lack of external validation limits the strength of construct validity claims for the trust, empathy, and accuracy facets specifically.</p><p>Fifth, as detailed in the Methods section, hallucination mitigation relied on architectural grounding and prompt-level guardrails&#x2014;an approach validated in the team&#x2019;s prior work rather than formal factual consistency evaluation or expert clinical review. The safety profile of HerCare&#x2019;s outputs specific to women&#x2019;s health topics remains to be established through independent clinical review. Additionally, this study&#x2019;s window overlapped with major Western holidays, which may have elevated stress levels and introduced temporal confounds for anxiety and mood-related topics, a contextual factor that cannot be fully disentangled from the results.</p><p>Sixth, the peer corpus carries inherent demographic biases: Reddit&#x2019;s user base skews younger, English-speaking, and predominantly Western, and score-based ranking may suppress minority perspectives. The five subreddits also underrepresent menopause, older women&#x2019;s fertility concerns, and non-Western postpartum experiences&#x2014;limitations that future studies should address through broader, more diverse corpus construction. Additionally, while transparent source attribution is a central design principle of HerCare, this study did not directly measure whether participants noticed or acted upon attribution labels; whether attribution meaningfully influenced trust calibration remains an open question for future evaluation.</p><p>Seventh, formal retrieval quality evaluation&#x2014;including relevance assessment, precision metrics, and ablation of retrieval parameters&#x2014;was not conducted; such benchmarking is recommended for future technical validation studies.</p></sec><sec id="s4-6"><title>Conclusions</title><p>This formative study suggests that the central tension in digital women&#x2019;s health AI between clinical accuracy and emotional resonance may be addressable through architectural design. By structuring the knowledge base as two distinct, purpose-built corpora and making their provenance visible in every response, HerCare was built to let users access vetted clinical guidance and lived community experience together rather than choosing between them; establishing whether this improves outcomes relative to alternatives will require controlled comparison. The most transferable contribution is the &#x201C;validate-then-redirect&#x201D; pattern characterized through the corpus analysis: a candidate design template for engineering empathy in health AI that can be instantiated through prompt engineering and retrieval architecture without fine-tuning or labeled emotional training data, and that warrants validation in future comparative work. More broadly, this work suggests that health AI should be evaluated not only on accuracy but on affective alignment, whether the system&#x2019;s linguistic behavior matches the emotional needs users bring to the interaction. The multimethod quantitative evaluation framework used here, combining standardized usability instruments with corpus-level computational linguistic analysis, offers a replicable methodology for assessing this alignment in future systems. Future work should pursue controlled comparative evaluation, expanded cultural and demographic diversity, formal clinical safety review, and longitudinal study of whether short-term trust and acceptance translate into sustained engagement and health behavior change.</p></sec></sec></body><back><ack><p>Regarding the use of generative AI (GenAI), the authors declare the use of GenAI in the research and writing process. According to the GAIDeT (2025; Generative Artificial Intelligence Delegation Taxonomy), the following tasks were delegated to GenAI tools under full human supervision: proofreading and editing, summarizing text, reformatting, and quality assessment. The GenAI tool used was Gemini (version 3; Google LLC). Responsibility for this final paper lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the outcomes.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the NSF (National Science Foundation; 2218046). The funder had no involvement in this study&#x2019;s design, data collection, analysis, interpretation, or the writing of this paper.</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during this study are not publicly available due to the sensitive nature of women&#x2019;s health discussions and to protect participant privacy in accordance with the Institutional Review Board&#x2013;approved data management protocol (Protocol #IRB0005368). The Python code used for the retrieval-augmented generation pipeline, computational analysis (latent Dirichlet allocation, VADER (Valence Aware Dictionary and Sentiment Reasoner), and National Research Council), and deidentified conversational data extracts have been archived in a secure, version-controlled repository. Access will be provided by the corresponding author upon reasonable request and subject to a data sharing agreement.</p></sec></notes><fn-group><fn fn-type="con"><p>KTZ was responsible for conceptualization, data curation, formal analysis, investigation, methodology, software, writing the original draft, and review and editing. WUH contributed to data curation, formal analysis, methodology, software, validation, visualization, writing the original draft, and review and editing. JL was responsible for funding acquisition, project administration, resources, supervision, validation, writing the original draft, and review and editing. NA contributed to conceptualization, methodology, project administration, supervision, validation, writing the original draft, and review and editing.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CA</term><def><p>conversational agent</p></def></def-item><def-item><term id="abb2">CUQ</term><def><p>Chatbot Usability Questionnaire</p></def></def-item><def-item><term id="abb3">EmoLex</term><def><p>Emotion Lexicon</p></def></def-item><def-item><term id="abb4">FAISS</term><def><p>Facebook AI Similarity Search</p></def></def-item><def-item><term id="abb5">FEC</term><def><p>focus areas competitive</p></def></def-item><def-item><term id="abb6">HCI</term><def><p>human-computer interaction</p></def></def-item><def-item><term id="abb7">LDA</term><def><p>latent Dirichlet allocation</p></def></def-item><def-item><term id="abb8">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb9">NPS</term><def><p>net promoter score</p></def></def-item><def-item><term id="abb10">NRC</term><def><p>National Research Council</p></def></def-item><def-item><term id="abb11">NSF</term><def><p>National Science Foundation</p></def></def-item><def-item><term id="abb12">PCOS</term><def><p>polycystic ovary syndrome</p></def></def-item><def-item><term id="abb13">PIQ</term><def><p>postinteraction questionnaire</p></def></def-item><def-item><term id="abb14">PRAW</term><def><p>Python Reddit API Wrapper</p></def></def-item><def-item><term id="abb15">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb16">RII</term><def><p>research infrastructure improvement</p></def></def-item><def-item><term id="abb17">VADER</term><def><p>Valence Aware Dictionary and Sentiment Reasoner</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Burst</surname><given-names>EL</given-names> </name><name name-style="western"><surname>Felice</surname><given-names>MC</given-names> </name><name name-style="western"><surname>O&#x2019;Kane</surname><given-names>AA</given-names> </name></person-group><article-title>Using and appropriating technology to support the menopause journey in the UK</article-title><conf-name>CHI &#x2019;24: Proceedings of the 2024 CHI Conference on Human Factors in Computing Systems</conf-name><conf-date>May 11-16, 2026</conf-date><pub-id pub-id-type="doi">10.1145/3613904.3642694</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rampazzo</surname><given-names>F</given-names> </name><name name-style="western"><surname>Raybould</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rampazzo</surname><given-names>P</given-names> </name><name name-style="western"><surname>Barker</surname><given-names>R</given-names> </name><name name-style="western"><surname>Leasure</surname><given-names>D</given-names> </name></person-group><article-title>&#x201C;UPDATE: I&#x2019;m pregnant!&#x201D;: inferring global downloads and reasons for using menstrual tracking apps</article-title><source>Digit Health</source><year>2024</year><month>11</month><day>21</day><volume>10</volume><fpage>20552076241298315</fpage><pub-id pub-id-type="doi">10.1177/20552076241298315</pub-id><pub-id pub-id-type="medline">39582950</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ciolfi Felice</surname><given-names>M</given-names> </name><name name-style="western"><surname>S&#x00F8;ndergaard</surname><given-names>MLJ</given-names> </name><name name-style="western"><surname>Balaam</surname><given-names>M</given-names> </name></person-group><article-title>Analyzing user reviews of the first digital contraceptive: mixed methods study</article-title><source>J Med Internet Res</source><year>2023</year><month>11</month><day>14</day><volume>25</volume><issue>1</issue><fpage>e47131</fpage><pub-id pub-id-type="doi">10.2196/47131</pub-id><pub-id pub-id-type="medline">37962925</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brammall</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Hayman</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Harrison</surname><given-names>CL</given-names> </name></person-group><article-title>Pregnancy mobile app use: a survey of health information practices and quality awareness among pregnant women in Australia</article-title><source>Womens Health (Lond)</source><year>2024</year><volume>20</volume><fpage>17455057241281236</fpage><pub-id pub-id-type="doi">10.1177/17455057241281236</pub-id><pub-id pub-id-type="medline">39501651</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nissen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>SY</given-names> </name><name name-style="western"><surname>J&#x00E4;ger</surname><given-names>KM</given-names> </name><etal/></person-group><article-title>Smartphone pregnancy apps: systematic analysis of features, scientific guidance, commercialization, and user perception</article-title><source>BMC Pregnancy Childbirth</source><year>2024</year><month>11</month><day>25</day><volume>24</volume><issue>1</issue><fpage>782</fpage><pub-id pub-id-type="doi">10.1186/s12884-024-06959-1</pub-id><pub-id pub-id-type="medline">39587534</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ibrahim</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Panchpor</surname><given-names>P</given-names> </name><name name-style="western"><surname>Nurain</surname><given-names>N</given-names> </name><name name-style="western"><surname>Clawson</surname><given-names>J</given-names> </name></person-group><article-title>"Islamically, i am no longer on my period&#x201D;: a study of menstrual tracking in muslim women in the US</article-title><conf-name>CHI &#x2019;24: Proceedings of the 2024 CHI Conference on Human Factors in Computing Systems</conf-name><conf-date>May 11-16, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3613904.3642006</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hawkins</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Kreuter</surname><given-names>M</given-names> </name><name name-style="western"><surname>Resnicow</surname><given-names>K</given-names> </name><name name-style="western"><surname>Fishbein</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dijkstra</surname><given-names>A</given-names> </name></person-group><article-title>Understanding tailoring in communicating about health</article-title><source>Health Educ Res</source><year>2008</year><month>06</month><volume>23</volume><issue>3</issue><fpage>454</fpage><lpage>466</lpage><pub-id pub-id-type="doi">10.1093/her/cyn004</pub-id><pub-id pub-id-type="medline">18349033</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kreuter</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Wray</surname><given-names>RJ</given-names> </name></person-group><article-title>Tailored and targeted health communication: strategies for enhancing information relevance</article-title><source>Am J Health Behav</source><year>2003</year><volume>27 Suppl 3</volume><fpage>S227</fpage><lpage>S232</lpage><pub-id pub-id-type="doi">10.5993/ajhb.27.1.s3.6</pub-id><pub-id pub-id-type="medline">14672383</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Street</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Makoul</surname><given-names>G</given-names> </name><name name-style="western"><surname>Arora</surname><given-names>NK</given-names> </name><name name-style="western"><surname>Epstein</surname><given-names>RM</given-names> </name></person-group><article-title>How does communication heal? Pathways linking clinician-patient communication to health outcomes</article-title><source>Patient Educ Couns</source><year>2009</year><month>03</month><volume>74</volume><issue>3</issue><fpage>295</fpage><lpage>301</lpage><pub-id pub-id-type="doi">10.1016/j.pec.2008.11.015</pub-id><pub-id pub-id-type="medline">19150199</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Derksen</surname><given-names>F</given-names> </name><name name-style="western"><surname>Bensing</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lagro-Janssen</surname><given-names>A</given-names> </name></person-group><article-title>Effectiveness of empathy in general practice: a systematic review</article-title><source>Br J Gen Pract</source><year>2013</year><month>01</month><volume>63</volume><issue>606</issue><fpage>e76</fpage><lpage>84</lpage><pub-id pub-id-type="doi">10.3399/bjgp13X660814</pub-id><pub-id pub-id-type="medline">23336477</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bagalkot</surname><given-names>N</given-names> </name><name name-style="western"><surname>Akbar</surname><given-names>SZ</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Embodied negotiations, practices and experiences interacting with pregnancy care infrastructures in south india</article-title><conf-name>Conference on Human Factors in Computing Systems - Proceedings Association for Computing Machinery</conf-name><conf-date>Apr 30 to May 5, 2022</conf-date><pub-id pub-id-type="doi">10.1145/3491102.3501950</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Khan</surname><given-names>KL</given-names> </name><name name-style="western"><surname>Azhar</surname><given-names>F</given-names> </name></person-group><article-title>Unpacking the interface: the impact of design choices on users&#x2019; relationship with raaji</article-title><conf-name>Proceedings of the 42nd ACM International Conference on Design of Communication, SIGDOC</conf-name><conf-date>Oct 20-22, 2024</conf-date><conf-loc>Fairfax, VA</conf-loc><fpage>138</fpage><lpage>150</lpage><pub-id pub-id-type="doi">10.1145/3641237.3691662</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mughal</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Aamir</surname><given-names>S</given-names> </name><name name-style="western"><surname>Samad</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zehra</surname><given-names>U</given-names> </name><name name-style="western"><surname>Syed</surname><given-names>AA</given-names> </name></person-group><article-title>Mai: a transformer based domain specific chatbot for menstrual health</article-title><source>ACM J Responsib Comput</source><year>2025</year><month>2</month><day>14</day><volume>2</volume><issue>1</issue><fpage>1</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1145/3711711</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sainz</surname><given-names>AN</given-names> </name><name name-style="western"><surname>Prabhakar</surname><given-names>A</given-names> </name></person-group><article-title>Women&#x2019;s health digital interventions in Latin America</article-title><year>2021</year><month>11</month><day>22</day><conf-name>CLIHC 2021</conf-name><conf-date>Nov 22-23, 2021</conf-date><pub-id pub-id-type="doi">10.1145/3488392.3488406</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rahman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rahman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tripto</surname><given-names>NI</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Apon</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Shahriyar</surname><given-names>R</given-names> </name></person-group><article-title>AdolescentBot: understanding opportunities for chatbots in combating adolescent sexual and reproductive health problems in Bangladesh</article-title><conf-name>CHI &#x2019;21: Proceedings of the 2021 CHI Conference on Human Factors in Computing Systems [Webinar]</conf-name><conf-date>May 8-13, 2021</conf-date><pub-id pub-id-type="doi">10.1145/3411764.3445694</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Upadhyay</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shekhawat</surname><given-names>M</given-names> </name><name name-style="western"><surname>Manhas</surname><given-names>R</given-names> </name></person-group><article-title>Data driven UX/UI design for reproductive health tracker</article-title><conf-name>2023 IEEE International Students&#x2019; Conference on Electrical, Electronics and Computer Science (SCEECS)</conf-name><conf-date>Feb 18-19, 2023</conf-date><pub-id pub-id-type="doi">10.1109/SCEECS57921.2023.10063134</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Verdezoto</surname><given-names>N</given-names> </name><name name-style="western"><surname>Carpio-Arias</surname><given-names>F</given-names> </name><name name-style="western"><surname>Carpio-Arias</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Indigenous women managing pregnancy complications in rural ecuador barriers and opportunities to enhance antenatal care</article-title><conf-name>NordiCHI &#x2019;20: Proceedings of the 11th Nordic Conference on Human-Computer Interaction: Shaping Experiences, Shaping Society [Webinar]</conf-name><conf-date>Oct 25-29, 2020</conf-date><pub-id pub-id-type="doi">10.1145/3419249.3420141</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>X</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>R</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>Y</given-names> </name></person-group><article-title>Perceived community support, users&#x2019; interactions, and value co-creation in online health community: the moderating effect of social exclusion</article-title><source>Int J Environ Res Public Health</source><year>2019</year><month>12</month><day>27</day><volume>17</volume><issue>1</issue><fpage>204</fpage><pub-id pub-id-type="doi">10.3390/ijerph17010204</pub-id><pub-id pub-id-type="medline">31892188</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Giorgi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Aich</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The illusion of empathy: how AI chatbots shape conversation perception</article-title><source>arXiv</source><access-date>2026-07-07</access-date><comment>Preprint posted online on  Nov 19, 2024</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/pdf/2411.12877v1">https://arxiv.org/pdf/2411.12877v1</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sanjeewa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Iyer</surname><given-names>R</given-names> </name><name name-style="western"><surname>Apputhurai</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wickramasinghe</surname><given-names>N</given-names> </name><name name-style="western"><surname>Meyer</surname><given-names>D</given-names> </name></person-group><article-title>Empathic conversational agent platform designs and their evaluation in the context of mental health: systematic review</article-title><source>JMIR Ment Health</source><year>2024</year><month>09</month><day>9</day><volume>11</volume><issue>1</issue><fpage>e58974</fpage><pub-id pub-id-type="doi">10.2196/58974</pub-id><pub-id pub-id-type="medline">39250799</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bizzotto</surname><given-names>N</given-names> </name><name name-style="western"><surname>Schulz</surname><given-names>PJ</given-names> </name><name name-style="western"><surname>de Bruijn</surname><given-names>GJ</given-names> </name></person-group><article-title>The &#x201C;Loci&#x201D; of misinformation and its correction in peer- and expert-led online communities for mental health: content analysis</article-title><source>J Med Internet Res</source><year>2023</year><month>09</month><day>18</day><volume>25</volume><fpage>e44656</fpage><pub-id pub-id-type="doi">10.2196/44656</pub-id><pub-id pub-id-type="medline">37721800</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Cuadra</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stein</surname><given-names>LA</given-names> </name><etal/></person-group><article-title>The illusion of empathy? Notes on displays of emotion in human-computer interaction</article-title><conf-name>CHI &#x2019;24: Proceedings of the 2024 CHI Conference on Human Factors in Computing Syste</conf-name><conf-date>May 11-16, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3613904.3642336</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lusi</surname><given-names>B</given-names> </name><name name-style="western"><surname>Petterson</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Thiruvenkatanathan</surname><given-names>KP</given-names> </name><etal/></person-group><article-title>Caring for reproductive justice: design in response to adversity</article-title><year>2024</year><month>11</month><day>11</day><conf-name>CSCW &#x2019;24</conf-name><conf-date>Nov 9-13, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3678884.3681832</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kumar</surname><given-names>N</given-names> </name><name name-style="western"><surname>Karusala</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ismail</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tuli</surname><given-names>A</given-names> </name></person-group><article-title>Taking the long, holistic, and intersectional view to women&#x2019;s wellbeing</article-title><source>ACM Trans Comput-Hum Interact</source><year>2020</year><month>7</month><day>20</day><volume>27</volume><issue>4</issue><fpage>1</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1145/3397159</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>S</given-names> </name><name name-style="western"><surname>Singhal</surname><given-names>A</given-names> </name><etal/></person-group><article-title>An artificial intelligence chatbot for young people&#x2019;s sexual and reproductive health in India (SnehAI): instrumental case study</article-title><source>J Med Internet Res</source><year>2022</year><month>01</month><day>3</day><volume>24</volume><issue>1</issue><fpage>e29969</fpage><pub-id pub-id-type="doi">10.2196/29969</pub-id><pub-id pub-id-type="medline">34982034</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name></person-group><article-title>Improving large language model applications in the medical and nursing domains with retrieval-augmented generation: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>10</month><day>21</day><volume>27</volume><issue>1</issue><fpage>e80557</fpage><pub-id pub-id-type="doi">10.2196/80557</pub-id><pub-id pub-id-type="medline">41118646</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abo El-Enen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Saad</surname><given-names>S</given-names> </name><name name-style="western"><surname>Nazmy</surname><given-names>T</given-names> </name></person-group><article-title>A survey on retrieval-augmentation generation (RAG) models for healthcare applications</article-title><source>Neural Comput Appl</source><year>2025</year><month>10</month><day>16</day><volume>37</volume><issue>33</issue><fpage>28191</fpage><lpage>28267</lpage><pub-id pub-id-type="doi">10.1007/s00521-025-11666-9</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Neha</surname><given-names>F</given-names> </name><name name-style="western"><surname>Bhati</surname><given-names>D</given-names> </name><name name-style="western"><surname>Shukla</surname><given-names>DK</given-names> </name></person-group><article-title>Retrieval-augmented generation (RAG) in healthcare: a comprehensive review</article-title><source>AI</source><year>2025</year><month>11</month><volume>6</volume><issue>9</issue><fpage>226</fpage><pub-id pub-id-type="doi">10.3390/ai6090226</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kao</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Chi</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>CY</given-names> </name></person-group><article-title>Exploring pregnant women&#x2019;s experiences with mobile chatbot-based antenatal education: a qualitative descriptive study</article-title><source>West J Nurs Res</source><year>2026</year><month>04</month><volume>48</volume><issue>4</issue><fpage>394</fpage><lpage>403</lpage><pub-id pub-id-type="doi">10.1177/01939459251408267</pub-id><pub-id pub-id-type="medline">41546497</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mills</surname><given-names>R</given-names> </name><name name-style="western"><surname>Mangone</surname><given-names>ER</given-names> </name><name name-style="western"><surname>Lesh</surname><given-names>N</given-names> </name><name name-style="western"><surname>Mohan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Baraitser</surname><given-names>P</given-names> </name></person-group><article-title>Chatbots to improve sexual and reproductive health: realist synthesis</article-title><source>J Med Internet Res</source><year>2023</year><month>08</month><day>9</day><volume>25</volume><fpage>e46761</fpage><pub-id pub-id-type="doi">10.2196/46761</pub-id><pub-id pub-id-type="medline">37556194</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Suharwardy</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ramachandran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Leonard</surname><given-names>SA</given-names> </name><etal/></person-group><article-title>Feasibility and impact of a mental health chatbot on postpartum mental health: a randomized controlled trial</article-title><source>AJOG Glob Rep</source><year>2023</year><month>03</month><day>29</day><volume>3</volume><issue>3</issue><fpage>100165</fpage><pub-id pub-id-type="doi">10.1016/j.xagr.2023.100165</pub-id><pub-id pub-id-type="medline">37560011</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Holmes</surname><given-names>S</given-names> </name><name name-style="western"><surname>Moorhead</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bond</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>H</given-names> </name><name name-style="western"><surname>Coates</surname><given-names>V</given-names> </name><name name-style="western"><surname>Mctear</surname><given-names>M</given-names> </name></person-group><article-title>Usability testing of a healthcare chatbot: can we use conventional methods to assess conversational user interfaces?</article-title><conf-name>ECCE &#x2019;19: Proceedings of the 31st European Conference on Cognitive Ergonomics</conf-name><conf-date>Sep 11-13, 2019</conf-date><conf-loc>Belfast, United Kingdom</conf-loc><fpage>207</fpage><lpage>214</lpage><pub-id pub-id-type="doi">10.1145/3335082.3335094</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singelis</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Garcia</surname><given-names>RI</given-names> </name><name name-style="western"><surname>Barker</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Davis</surname><given-names>RE</given-names> </name></person-group><article-title>An experimental test of the two-dimensional theory of cultural sensitivity in health communication</article-title><source>J Health Commun</source><year>2018</year><volume>23</volume><issue>4</issue><fpage>321</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1080/10810730.2018.1443526</pub-id><pub-id pub-id-type="medline">29509068</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parvanta</surname><given-names>C</given-names> </name><name name-style="western"><surname>Murray</surname><given-names>K</given-names> </name><name name-style="western"><surname>Boddupalli</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Development of the cultural tailoring score (CTS): a scoring instrument to assess cultural tailoring of health messaging</article-title><source>Health Lit Commun Open</source><year>2025</year><month>12</month><day>31</day><volume>3</volume><issue>1</issue><fpage>2459199</fpage><pub-id pub-id-type="doi">10.1080/28355245.2025.2459199</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Laranjo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Dunn</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Tong</surname><given-names>HL</given-names> </name><etal/></person-group><article-title>Conversational agents in healthcare: a systematic review</article-title><source>J Am Med Inform Assoc</source><year>2018</year><month>09</month><day>1</day><volume>25</volume><issue>9</issue><fpage>1248</fpage><lpage>1258</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocy072</pub-id><pub-id pub-id-type="medline">30010941</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tam</surname><given-names>TYC</given-names> </name><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kapoor</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>28</day><volume>7</volume><issue>1</issue><fpage>258</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id><pub-id pub-id-type="medline">39333376</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ghassemi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>S</given-names> </name></person-group><article-title>Large language models in biomedicine and health: current research landscape and future directions</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>09</month><day>1</day><volume>31</volume><issue>9</issue><fpage>1801</fpage><lpage>1811</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae202</pub-id><pub-id pub-id-type="medline">39169867</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kazakoff</surname><given-names>A</given-names> </name><name name-style="western"><surname>Doroshuk</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Ganshorn</surname><given-names>H</given-names> </name><name name-style="western"><surname>Doyle-Baker</surname><given-names>PK</given-names> </name></person-group><article-title>Motivations for use, user experience and quality of reproductive health mobile applications in a pre-menopausal user base: a scoping review</article-title><source>Healthcare (Basel)</source><year>2025</year><month>04</month><day>11</day><volume>13</volume><issue>8</issue><fpage>877</fpage><pub-id pub-id-type="doi">10.3390/healthcare13080877</pub-id><pub-id pub-id-type="medline">40281826</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pichon</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jackman</surname><given-names>KB</given-names> </name><name name-style="western"><surname>Winkler</surname><given-names>IT</given-names> </name><name name-style="western"><surname>Bobel</surname><given-names>C</given-names> </name><name name-style="western"><surname>Elhadad</surname><given-names>N</given-names> </name></person-group><article-title>The messiness of the menstruator: assessing personas and functionalities of menstrual tracking apps</article-title><source>J Am Med Inform Assoc</source><year>2022</year><month>01</month><day>12</day><volume>29</volume><issue>2</issue><fpage>385</fpage><lpage>399</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocab212</pub-id><pub-id pub-id-type="medline">34613388</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hammond</surname><given-names>E</given-names> </name><name name-style="western"><surname>Burdon</surname><given-names>M</given-names> </name></person-group><article-title>Intimate harms and menstrual cycle tracking apps</article-title><source>Comput Law Secur Rev</source><year>2024</year><month>11</month><volume>55</volume><fpage>106038</fpage><pub-id pub-id-type="doi">10.1016/j.clsr.2024.106038</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sillence</surname><given-names>E</given-names> </name><name name-style="western"><surname>Osborne</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Kemp</surname><given-names>E</given-names> </name><name name-style="western"><surname>McKellar</surname><given-names>K</given-names> </name></person-group><article-title>Menopause apps: personal health tracking, empowerment and epistemic injustice</article-title><source>Digit Health</source><year>2025</year><month>04</month><day>27</day><volume>11</volume><fpage>20552076251330782</fpage><pub-id pub-id-type="doi">10.1177/20552076251330782</pub-id><pub-id pub-id-type="medline">40297381</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Epstein</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>NB</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>JH</given-names> </name><etal/></person-group><article-title>Examining menstrual tracking to inform the design of personal informatics tools</article-title><source>Proc SIGCHI Conf Hum Factor Comput Syst</source><year>2017</year><month>05</month><day>2</day><volume>2017</volume><fpage>6876</fpage><lpage>6888</lpage><pub-id pub-id-type="doi">10.1145/3025453.3025635</pub-id><pub-id pub-id-type="medline">28516176</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Malki</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Kaleva</surname><given-names>I</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>D</given-names> </name><name name-style="western"><surname>Warner</surname><given-names>M</given-names> </name><name name-style="western"><surname>Abu-Salma</surname><given-names>R</given-names> </name></person-group><article-title>Exploring privacy practices of female mhealth apps in a post-roe world</article-title><conf-name>Conference on Human Factors in Computing Systems - Proceedings Association for Computing Machinery</conf-name><conf-date>May 11-16, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3613904.3642521</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Cao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Laabadli</surname><given-names>H</given-names> </name><name name-style="western"><surname>Mathis</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Stern</surname><given-names>RD</given-names> </name><name name-style="western"><surname>Emami-Naeini</surname><given-names>P</given-names> </name></person-group><article-title>&#x201C;I deleted it after the overturn of roe v. wade&#x201D;: understanding women&#x2019;s privacy concerns toward period-tracking apps in the post Roe v. Wade era</article-title><conf-name>Conference on Human Factors in Computing Systems - Proceedings Association for Computing Machinery</conf-name><conf-date>May 11-16, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3613904.3642042</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Hernandez</surname><given-names>RH</given-names> </name><name name-style="western"><surname>Kou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gui</surname><given-names>X</given-names> </name></person-group><article-title>&#x201C;Our users&#x2019; privacy is paramount to us&#x201D;: a discourse analysis of how period and fertility tracking app companies address the roe v wade overturn</article-title><conf-name>Conference on Human Factors in Computing Systems - Proceedings Association for Computing Machinery</conf-name><conf-date>May 11-16, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3613904.3642384</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bucher</surname><given-names>E</given-names> </name><name name-style="western"><surname>Sharkey</surname><given-names>C</given-names> </name><name name-style="western"><surname>Henderson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Basaran</surname><given-names>B</given-names> </name><name name-style="western"><surname>Meyer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>CX</given-names> </name></person-group><article-title>An evaluation of menstrual health apps&#x2019; functionality, inclusiveness, and health education information</article-title><source>BMC Womens Health</source><year>2025</year><month>05</month><day>28</day><volume>25</volume><issue>1</issue><fpage>261</fpage><pub-id pub-id-type="doi">10.1186/s12905-025-03812-1</pub-id><pub-id pub-id-type="medline">40437458</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="thesis"><person-group person-group-type="author"><name name-style="western"><surname>Kukreja</surname><given-names>K</given-names> </name></person-group><article-title>Bridging the gap: reimagining menstrual tracking apps to meet the needs of women with PCOS</article-title><year>2025</year><access-date>2026-07-07</access-date><publisher-name>Tilburg University</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://arno.uvt.nl/show.cgi?fid=183009">https://arno.uvt.nl/show.cgi?fid=183009</ext-link></comment></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zadushlivy</surname><given-names>N</given-names> </name><name name-style="western"><surname>Biviji</surname><given-names>R</given-names> </name><name name-style="western"><surname>Williams</surname><given-names>KS</given-names> </name></person-group><article-title>Exploration of reproductive health apps&#x2019; data privacy policies and the risks posed to users: qualitative content analysis</article-title><source>J Med Internet Res</source><year>2025</year><month>03</month><day>5</day><volume>27</volume><fpage>e51517</fpage><pub-id pub-id-type="doi">10.2196/51517</pub-id><pub-id pub-id-type="medline">40053713</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Young</surname><given-names>J</given-names> </name><name name-style="western"><surname>Stacy</surname><given-names>P</given-names> </name><name name-style="western"><surname>Nadia</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Critiquing menstrual pain technologies through the lens of feminist disability studies</article-title><conf-name>Proceedings of the CHI Conference on Human Factors in Computing Systems (CHI &#x2019;24)</conf-name><conf-date>May 11, 2024 to May 16, 2025</conf-date><pub-id pub-id-type="doi">10.1145/3613904.3642691</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chowdhury</surname><given-names>HM</given-names> </name><name name-style="western"><surname>Sultana</surname><given-names>S</given-names> </name></person-group><article-title>From literature to &#x201C;rewa&#x201D;: discussing reproductive well-being in HCI</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 1, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.01121</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bickmore</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gruber</surname><given-names>A</given-names> </name><name name-style="western"><surname>Picard</surname><given-names>R</given-names> </name></person-group><article-title>Establishing the computer-patient working alliance in automated health behavior change interventions</article-title><source>Patient Educ Couns</source><year>2005</year><month>10</month><volume>59</volume><issue>1</issue><fpage>21</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.1016/j.pec.2004.09.008</pub-id><pub-id pub-id-type="medline">16198215</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bickmore</surname><given-names>TW</given-names> </name><name name-style="western"><surname>Picard</surname><given-names>RW</given-names> </name></person-group><article-title>Establishing and maintaining long-term human-computer relationships</article-title><source>ACM Trans Comput-Hum Interact</source><year>2005</year><month>06</month><volume>12</volume><issue>2</issue><fpage>293</fpage><lpage>327</lpage><pub-id pub-id-type="doi">10.1145/1067860.1067867</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Noar</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Benac</surname><given-names>CN</given-names> </name><name name-style="western"><surname>Harris</surname><given-names>MS</given-names> </name></person-group><article-title>Does tailoring matter? Meta-analytic review of tailored print health behavior change interventions</article-title><source>Psychol Bull</source><year>2007</year><month>07</month><volume>133</volume><issue>4</issue><fpage>673</fpage><lpage>693</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.133.4.673</pub-id><pub-id pub-id-type="medline">17592961</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lustria</surname><given-names>MLA</given-names> </name><name name-style="western"><surname>Noar</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Cortese</surname><given-names>J</given-names> </name><name name-style="western"><surname>Van Stee</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Glueckauf</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name></person-group><article-title>A meta-analysis of web-delivered tailored health behavior change interventions</article-title><source>J Health Commun</source><year>2013</year><volume>18</volume><issue>9</issue><fpage>1039</fpage><lpage>1069</lpage><pub-id pub-id-type="doi">10.1080/10810730.2013.768727</pub-id><pub-id pub-id-type="medline">23750972</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mustafa</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zaman</surname><given-names>KT</given-names> </name><name name-style="western"><surname>Ahmad</surname><given-names>T</given-names> </name><name name-style="western"><surname>Batool</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ghazali</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>N</given-names> </name></person-group><article-title>Religion and women&#x2019;s intimate health: towards an inclusive approach to healthcare</article-title><conf-name>Conference on Human Factors in Computing Systems - Proceedings Association for Computing Machinery</conf-name><conf-date>May 8-13, 2021</conf-date><pub-id pub-id-type="doi">10.1145/3411764.3445605</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Al-Naimi</surname><given-names>L</given-names> </name><name name-style="western"><surname>Alistar</surname><given-names>M</given-names> </name></person-group><article-title>Understanding cultural and religious values relating to awareness of women&#x2019;s intimate health among arab muslims</article-title><conf-name>CHI &#x2019;24: Proceedings of the 2024 CHI Conference on Human Factors in Computing Systems</conf-name><conf-date>May 11-16, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3613904.3642207</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sultana</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chowdhury</surname><given-names>HM</given-names> </name><name name-style="western"><surname>Sultana</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Verdezoto</surname><given-names>N</given-names> </name></person-group><article-title>&#x201C;Socheton&#x201D;: a culturally appropriate AI tool to support reproductive well-being</article-title><conf-name>DIS &#x2019;25: Proceedings of the 2025 ACM Designing Interactive Systems Conference</conf-name><conf-date>Jul 5-9, 2025</conf-date><pub-id pub-id-type="doi">10.1145/3715336.3735725</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zaman</surname><given-names>KT</given-names> </name><name name-style="western"><surname>Hasan</surname><given-names>WU</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name></person-group><article-title>Exploring the dynamics of online social support for ADRD caregivers: a study on online peer support groups</article-title><conf-name>2023 IEEE International Conference on E-health Networking, Application &#x0026; Services (Healthcom)</conf-name><conf-date>Dec 15-17, 2023</conf-date><pub-id pub-id-type="doi">10.1109/Healthcom56612.2023.10472395</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naseem</surname><given-names>M</given-names> </name><name name-style="western"><surname>Younas</surname><given-names>F</given-names> </name><name name-style="western"><surname>Mustafa</surname><given-names>M</given-names> </name></person-group><article-title>Designing digital safe spaces for peer support and connectivity in patriarchal contexts</article-title><source>Proc ACM Hum-Comput Interact</source><year>2020</year><month>10</month><day>14</day><volume>4</volume><issue>CSCW2</issue><fpage>1</fpage><lpage>24</lpage><pub-id pub-id-type="doi">10.1145/3415217</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ammari</surname><given-names>T</given-names> </name><name name-style="western"><surname>Nofal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Naseem</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mustafa</surname><given-names>M</given-names> </name></person-group><article-title>Moderation as empowerment: creating and managing women-only digital safe spaces</article-title><source>Proc ACM Hum-Comput Interact</source><year>2022</year><month>11</month><day>7</day><volume>6</volume><issue>CSCW2</issue><fpage>1</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.1145/3555204</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tam</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bhat</surname><given-names>KS</given-names> </name><name name-style="western"><surname>Mohindra</surname><given-names>P</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>N</given-names> </name></person-group><article-title>Learning to navigate health taboos through online safe spaces</article-title><conf-name>Conference on Human Factors in Computing Systems - Proceedings Association for Computing Machinery</conf-name><conf-date>Apr 23-28, 2023</conf-date><pub-id pub-id-type="doi">10.1145/3544548.3580708</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miller</surname><given-names>B</given-names> </name></person-group><article-title>View of investigating Reddit self-disclosure and confessions in relation to connectedness, social support, and life satisfaction</article-title><source>JSMS</source><year>2020</year><access-date>2026-07-24</access-date><volume>9</volume><issue>1</issue><fpage>39</fpage><lpage>62</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://thejsms.org/index.php/JSMS/article/view/535/335">https://thejsms.org/index.php/JSMS/article/view/535/335</ext-link></comment></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Counts</surname><given-names>S</given-names> </name><name name-style="western"><surname>Horvitz</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Hoff</surname><given-names>A</given-names> </name><name name-style="western"><surname>Choudhury</surname><given-names>MD</given-names> </name></person-group><article-title>Characterizing and predicting postpartum depression from shared Facebook data</article-title><conf-name>Proceedings of the ACM Conference on Computer Supported Cooperative Work, CSCW</conf-name><conf-date>Feb 15-19, 2014</conf-date><pub-id pub-id-type="doi">10.1145/2531602.2531675</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Andalibi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Haimson</surname><given-names>OL</given-names> </name><name name-style="western"><surname>Choudhury</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Forte</surname><given-names>A</given-names> </name></person-group><article-title>Social support, reciprocity, and anonymity in responses to sexual abuse disclosures on social media</article-title><source>ACM Trans Comput-Hum Interact</source><year>2018</year><month>10</month><day>31</day><volume>25</volume><issue>5</issue><fpage>1</fpage><lpage>35</lpage><pub-id pub-id-type="doi">10.1145/3234942</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Hidden flaws behind expert-level accuracy of multimodal GPT-4 vision in medicine</article-title><source>NPJ Digit Med</source><year>2024</year><month>07</month><day>23</day><volume>7</volume><issue>1</issue><fpage>190</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01185-7</pub-id><pub-id pub-id-type="medline">39043988</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Nori</surname><given-names>H</given-names> </name><name name-style="western"><surname>King</surname><given-names>N</given-names> </name><name name-style="western"><surname>Mckinney</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Carignan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Horvitz</surname><given-names>E</given-names> </name></person-group><article-title>Capabilities of GPT-4 on medical challenge problems</article-title><source>arXiv</source><access-date>2026-07-07</access-date><comment>Preprint posted online on  Apr 12, 2023</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/pdf/2303.13375">https://arxiv.org/pdf/2303.13375</ext-link></comment></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arslan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ghanem</surname><given-names>H</given-names> </name><name name-style="western"><surname>Munawar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cruz</surname><given-names>C</given-names> </name></person-group><article-title>A survey on RAG with LLMs</article-title><source>Procedia Comput Sci</source><year>2024</year><volume>246</volume><fpage>3781</fpage><lpage>3790</lpage><pub-id pub-id-type="doi">10.1016/j.procs.2024.09.178</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for large language models: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 27, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2312.10997</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>P</given-names> </name><name name-style="western"><surname>Perez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Piktus</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for knowledge-intensive NLP tasks</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 12, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2005.11401</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jia</surname><given-names>N</given-names> </name><name name-style="western"><surname>Wakslak</surname><given-names>CJ</given-names> </name></person-group><article-title>AI can help people feel heard, but an AI label diminishes this impact</article-title><source>Proc Natl Acad Sci U S A</source><year>2024</year><month>04</month><day>2</day><volume>121</volume><issue>14</issue><fpage>e2319112121</fpage><pub-id pub-id-type="doi">10.1073/pnas.2319112121</pub-id><pub-id pub-id-type="medline">38551835</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="web"><article-title>Labeling AI-generated content may not change its persuasiveness</article-title><source>Stanford University Human-Centered Artificial Intelligence</source><access-date>2026-07-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://hai.stanford.edu/policy/labeling-ai-generated-content-may-not-change-its-persuasiveness">https://hai.stanford.edu/policy/labeling-ai-generated-content-may-not-change-its-persuasiveness</ext-link></comment></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shen</surname><given-names>J</given-names> </name><name name-style="western"><surname>DiPaola</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sap</surname><given-names>M</given-names> </name><name name-style="western"><surname>Park</surname><given-names>HW</given-names> </name><name name-style="western"><surname>Breazeal</surname><given-names>C</given-names> </name></person-group><article-title>Empathy toward artificial intelligence versus human experiences and the role of transparency in mental health and social support chatbot design: comparative study</article-title><source>JMIR Ment Health</source><year>2024</year><month>09</month><day>25</day><volume>11</volume><issue>1</issue><fpage>e62679</fpage><pub-id pub-id-type="doi">10.2196/62679</pub-id><pub-id pub-id-type="medline">39321450</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>System implementation details.</p><media xlink:href="formative_v10i1e88549_app1.docx" xlink:title="DOCX File, 29 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Recruitment materials and informed consent.</p><media xlink:href="formative_v10i1e88549_app2.docx" xlink:title="DOCX File, 24 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Supplementary computational analysis data.</p><media xlink:href="formative_v10i1e88549_app3.docx" xlink:title="DOCX File, 27 KB"/></supplementary-material></app-group></back></article>