<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Res Protoc</journal-id><journal-id journal-id-type="publisher-id">ResProt</journal-id><journal-id journal-id-type="index">5</journal-id><journal-title>JMIR Research Protocols</journal-title><abbrev-journal-title>JMIR Res Protoc</abbrev-journal-title><issn pub-type="epub">1929-0748</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v15i1e79966</article-id><article-id pub-id-type="doi">10.2196/79966</article-id><article-categories><subj-group subj-group-type="heading"><subject>Protocol</subject></subj-group></article-categories><title-group><article-title>A Conversational Agent for Providing Personalized Preexposure Prophylaxis (PrEP) Support: Protocol for Chatbot Implementation and Evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Sayed</surname><given-names>Fatima</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Park</surname><given-names>Albert</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sullivan</surname><given-names>Patrick Sean</given-names></name><degrees>DVM, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ge</surname><given-names>Yaorong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Software and Information Systems, University of North Carolina at Charlotte</institution><addr-line>Charlotte</addr-line><addr-line>NC</addr-line><country>United States</country></aff><aff id="aff2"><institution>School of Health Information Science, University of Victoria</institution><addr-line>PO Box 1700 STN CSC</addr-line><addr-line>Victoria</addr-line><addr-line>BC</addr-line><country>Canada</country></aff><aff id="aff3"><institution>Department of Epidemiology, Emory University</institution><addr-line>Atlanta</addr-line><addr-line>GA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Chatzimina</surname><given-names>Maria</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Ha</surname><given-names>Sook</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Albert Park, PhD, School of Health Information Science, University of Victoria, PO Box 1700 STN CSC, Victoria, BC, V8W 2Y2, Canada, +1-250-721-8575; <email>alpark@uvic.ca</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>1</day><month>10</month><year>2026</year></pub-date><volume>15</volume><elocation-id>e79966</elocation-id><history><date date-type="received"><day>01</day><month>07</month><year>2025</year></date><date date-type="rev-recd"><day>02</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>03</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Fatima Sayed, Albert Park, Patrick Sean Sullivan, Yaorong Ge. Originally published in JMIR Research Protocols (<ext-link ext-link-type="uri" xlink:href="https://www.researchprotocols.org">https://www.researchprotocols.org</ext-link>), 1.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Research Protocols, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.researchprotocols.org">https://www.researchprotocols.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.researchprotocols.org/2026/1/e79966"/><abstract><sec><title>Background</title><p>Chatbots have the potential to reduce barriers to preexposure prophylaxis (PrEP), including lack of awareness, misconceptions, and stigma, by providing anonymous and continuous support. However, in the context of PrEP, chatbots are still nascent; they lack personalized informational expertise, peer experiential expertise, and human-like emotional support to facilitate future PrEP uptake. These personalized and relatable forms of support are crucial for increasing engagement, influencing health decisions, and fostering resilience and well-being.</p></sec><sec><title>Objective</title><p>In this paper, we describe the iterative development and evaluation plans of a retrieval-augmented generation (RAG) chatbot for providing personalized information, peer experiential expertise, and human-like emotional support to PrEP candidates.</p></sec><sec sec-type="methods"><title>Methods</title><p>We used an iterative design process consisting of 2 phases: prototype conceptualization and iterative chatbot development. In the conceptualization phase, we identified real-world PrEP needs and designed a functional dialogue flow diagram for PrEP support. Chatbot development included developing 2 components: a query preprocessor and a RAG module. The preprocessor uses the Segment Any Text (SAT) tool (developed by Markus Frohmann, Igor Sterner, Ivan Vuli&#x0107;, Benjamin Minixhofer, and Markus Schedl) for query segmentation and a Gemma 2 fine-tuned support classifier to identify informational, emotional, and contextual data from real-world queries. To implement the RAG module, we used Sentence-Bidirectional Encoder Representations from Transformers (SBERT) embeddings with cosine similarity, and performed topic matching to identify topically relevant documents based on the query topic to support document retrieval. Extensive prompt engineering was used to guide the large language model (LLM), Gemini-2.0-flash, in generating tailored responses. We conducted 10 rounds of internal evaluations to assess and improve the chatbot responses based on 10 criteria: clarity, accuracy, actionability, relevancy, information detail, tailored information, comprehensiveness, language suitability, tone, and empathy. Finally, we conducted a blinded comparative study and used a linear mixed model to validate the chatbot against a general LLM and real-world user responses.</p></sec><sec sec-type="results"><title>Results</title><p>We developed a RAG chatbot and iteratively refined it based on the internal evaluation feedback. Prompt engineering is essential in guiding the LLM to generate responses tailored to information, experiential, and emotional user needs. We found that prompt effectiveness varied with task complexity; this was likely due to LLM sensitivity to the structure of prompts and to linguistic variability. Prompt decomposition and segmenting prompt instructions helped improve comprehensiveness and relevancy for complex and long queries. The linear mixed model analysis revealed that while both the general LLM and RAG were preferred over user responses, the general LLM maintained a superior edge over the RAG chatbot across most criteria, with the exceptions of empathy and tone.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Our RAG chatbot leverages social media data to provide personalized information, peer experiences, and human-like emotional support; these elements are essential for addressing PrEP misconceptions and promoting self-efficacy. Further analysis incorporating expert and user feedback will be conducted to help validate and improve the chatbot&#x2019;s potential.</p></sec><sec sec-type="registered-report"><title>International Registered Report Identifier (IRRID)</title><p>PRR1-10.2196/79966</p></sec></abstract><kwd-group><kwd>HIV prevention</kwd><kwd>chatbots</kwd><kwd>conversational agents</kwd><kwd>large language models</kwd><kwd>retrieval-augmented generation</kwd><kwd>generative AI</kwd><kwd>PrEP</kwd><kwd>preexposure prophylaxis</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Increasing the uptake of preexposure prophylaxis (PrEP) [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>] is a key strategy in the Ending the HIV Epidemic (EHE) initiative [<xref ref-type="bibr" rid="ref4">4</xref>]. However, social and structural barriers to HIV treatment-seeking behaviors continue to impact health outcomes and drive inequities [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. PrEP adoption has been hindered in the United States by lack of PrEP knowledge and prevailing anticipated and experienced stigma and discrimination [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. Personal experiences shared by peers can improve the awareness [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>], self-efficacy, and intrinsic motivation [<xref ref-type="bibr" rid="ref12">12</xref>] of people considering PrEP use, mitigating their stigmatized beliefs [<xref ref-type="bibr" rid="ref10">10</xref>], and informing health decisions [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. People without financial means or ready access to providers experience limited access to health care [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Service deserts result in a lack of physically proximate service providers [<xref ref-type="bibr" rid="ref6">6</xref>]; scarcity of services also disproportionately impacts those without financial means or access [<xref ref-type="bibr" rid="ref16">16</xref>]. Equitable PrEP uptake requires facilitating access to PrEP resources while meeting the diverse needs of PrEP candidates [<xref ref-type="bibr" rid="ref16">16</xref>]. Conversational agents (ie, chatbots) have the potential to provide anonymous, round-the-clock assistance in finding a PrEP provider, answering questions about PrEP, and providing emotional support. These services can help to overcome geographical and access barriers to seeking PrEP care.</p><p>Integrating chatbots into existing public health programs can significantly mitigate treatment barriers while enhancing user trust and adoption. However, to ensure scalability, deployment must account for infrastructure constraints and linguistic diversity. In low-resource settings where internet access, device affordability, and digital literacy may be limited [<xref ref-type="bibr" rid="ref17">17</xref>], chatbots should be designed for local deployment to ensure data privacy and accessibility. This includes supporting offline functionality for mobile devices and using lightweight architectures to ensure accessibility on older hardware. Furthermore, to cater to cultural and behavioral preferences, it is essential to incorporate multilingual and multiplatform support. Providing such user-centric personalization is a vital step in reducing health disparities and ensuring that digital interventions are both accessible and equitable.</p></sec><sec id="s1-2"><title>HIV and PrEP Chatbots</title><p>Chatbots have demonstrated potential in disseminating HIV information [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>], self-testing [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>], access to and uptake of HIV prevention and care [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref24">24</xref>], and making positive behavioral changes relevant to HIV prevention and care (eg, self-disclosure of HIV status [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>] and self-management of PrEP adherence [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]). AI technologies can be useful in developing and operating chatbot services, but their contribution is variable depending on the risk and complexity associated with health infrastructure. This is because risk and complexity influence the adoption of chatbots by users for improving health conditions [<xref ref-type="bibr" rid="ref28">28</xref>]. Rule-based chatbots, which use scripted text based on simple rules (ie, keyword identification and pattern-matching techniques), are most commonly used to provide information [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref29">29</xref>] and support self-management of health behaviors (eg, setting up reminders) [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. More advanced chatbots use information retrieval systems [<xref ref-type="bibr" rid="ref23">23</xref>] and knowledge graphs [<xref ref-type="bibr" rid="ref21">21</xref>] to help users identify HIV risk and navigate relevant information to mitigate barriers to PrEP uptake. For more complex needs such as identifying disorders and providing behavioral support (eg, for PrEP retention), chatbots use hybrid techniques that use AI tools (eg, machine learning [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>] and natural language processing [<xref ref-type="bibr" rid="ref21">21</xref>]) to learn patterns from previous conversations [<xref ref-type="bibr" rid="ref30">30</xref>]. These approaches can also detect emotions [<xref ref-type="bibr" rid="ref30">30</xref>] and perceive user characteristics based on past interactions [<xref ref-type="bibr" rid="ref21">21</xref>], and can integrate dialogue rules for response generation. Two protocols for studies mention the use of a hybrid large language model (LLM; HumanX) chatbot for promoting PrEP uptake and use [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>] and for providing HIV-related mental health support [<xref ref-type="bibr" rid="ref22">22</xref>]. However, these studies do not specify details of the chatbot implementation. Such rule-based and hybrid health care chatbots are often criticized for their limited flexibility [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>] and diversity of content [<xref ref-type="bibr" rid="ref31">31</xref>-<xref ref-type="bibr" rid="ref36">36</xref>], and insufficient contextual understanding [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>] and personalization [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. Such chatbots have also been reported to lack human-like emotional support [<xref ref-type="bibr" rid="ref40">40</xref>], and this limits long-term user engagement and chatbot effectiveness [<xref ref-type="bibr" rid="ref41">41</xref>]. LLMs, leveraging vast knowledge bases, can generate responses that are varied and context-relevant compared to rule-based and hybrid systems [<xref ref-type="bibr" rid="ref42">42</xref>]. Beyond information retrieval, LLMs have demonstrated a significant capacity for cognitive empathy, the intellectual ability to identify and label a user&#x2019;s emotional state and respond with supportive language [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. However, researchers have noted that these responses often lack affective empathy [<xref ref-type="bibr" rid="ref44">44</xref>], as the models lack human-like emotional experience and intrinsic motivation [<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], leading to impersonal responses in sensitive and stigma-associated contexts [<xref ref-type="bibr" rid="ref45">45</xref>]. This empathy gap suggests that while LLMs can extend general warmth, they struggle to provide the deep, personalized emotional connection found in human-to-human emotional support (ie, human-like emotional support).</p><p>To date, no study details the implementation of an LLM-based chatbot for providing various HIV and PrEP support, including emotional, informational, and peer experiential support to facilitate access to PrEP information and promote PrEP uptake and adherence. Although LLMs can offer improved personalization and human-like conversations [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref46">46</xref>], their performance can also be limited by their scope of specialized knowledge, diversity of available data, and the potential for LLM hallucinations [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>]. Fine-tuned LLMs have been effective in performing domain-specific tasks; however, the small size of the training corpus limits their ability to generalize model capabilities across diverse user needs [<xref ref-type="bibr" rid="ref31">31</xref>]. Prompt engineering [<xref ref-type="bibr" rid="ref49">49</xref>] and retrieval-augmented generation (RAG) [<xref ref-type="bibr" rid="ref48">48</xref>] techniques improve LLMs&#x2019; performance in terms of accuracy [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>], relevancy [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>], data privacy [<xref ref-type="bibr" rid="ref53">53</xref>], and user personalization [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref55">55</xref>], which can enhance user trust in chatbot-provided support [<xref ref-type="bibr" rid="ref56">56</xref>-<xref ref-type="bibr" rid="ref58">58</xref>]. Using prompt engineering, LLMs can be guided to improve the precision and accuracy of generated responses [<xref ref-type="bibr" rid="ref57">57</xref>]. RAG chatbots have indicated improved accuracy (34%) [<xref ref-type="bibr" rid="ref49">49</xref>] and reduced hallucination (26.5%) [<xref ref-type="bibr" rid="ref48">48</xref>] in generated responses compared to LLM question-answering bots. Compared to fine-tuned LLMs, RAG provides improved quality of responses in specialized domains by leveraging diverse external knowledge [<xref ref-type="bibr" rid="ref27">27</xref>]. Using effective prompts to guide LLMs [<xref ref-type="bibr" rid="ref32">32</xref>], RAG chatbots can offer greater personalization and relevance compared to rule-based and hybrid systems [<xref ref-type="bibr" rid="ref27">27</xref>].</p><p>However, the development of chatbots based on LLM and RAG techniques is still nascent in the context of HIV and PrEP support and has not been implemented to address implicit user needs [<xref ref-type="bibr" rid="ref48">48</xref>], answer long contextual queries [<xref ref-type="bibr" rid="ref48">48</xref>], and generate personalized responses [<xref ref-type="bibr" rid="ref59">59</xref>]. Addressing these issues in RAG chatbots can lead to providing accurate, flexible, personalized, and human-like responses to promote PrEP uptake.</p><p>Available literature [<xref ref-type="bibr" rid="ref60">60</xref>-<xref ref-type="bibr" rid="ref62">62</xref>] and our prior work have identified human-like emotional support components as essential elements to support behavior change for HIV medication taking and PrEP uptake. However, no prior work has developed RAG chatbots to provide personalized information and peer experiential expertise (eg, user stories) [<xref ref-type="bibr" rid="ref63">63</xref>]. Existing HIV and PrEP chatbots focus on providing informational support [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref29">29</xref>] and emotional support according to therapeutic guidelines [<xref ref-type="bibr" rid="ref27">27</xref>]. Effective support provided by humans relies on emotional support (ie, human-like emotional support) [<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref65">65</xref>] and peer experiential expertise (ie, peer experiential support) [<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref66">66</xref>], elements which are absent in existing health care chatbots. Providing effective human-like emotional support first requires understanding and communicating complex human emotional experiences [<xref ref-type="bibr" rid="ref67">67</xref>-<xref ref-type="bibr" rid="ref69">69</xref>]. The Willcox Feeling Wheel facilitates this process by providing a related vocabulary for 6 emotion categories [<xref ref-type="bibr" rid="ref69">69</xref>]. To better reflect the real-world user queries that often seek practical user experiences (peer experiential expertise) [<xref ref-type="bibr" rid="ref70">70</xref>], PrEP chatbots should provide peer experiential expertise support through pragmatic narratives of others in similar situations, which might enhance self-coping skills and emotional confidence, leading to improved quality of life [<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref66">66</xref>]. Additionally, chatbots need to support back-and-forth dialogue exchanges, similar to real-world help-seeking behaviors, to facilitate clarification and understanding of complex concerns [<xref ref-type="bibr" rid="ref71">71</xref>].</p><p>We develop a RAG chatbot that addresses the knowledge gaps in providing personalized PrEP support. We develop a support classifier to identify implicit PrEP user needs from long context queries by fine-tuning Google&#x2019;s Gemma model. The RAG chatbot uses personalized prompts and an external knowledge base (facts and real-world PrEP experiences) to generate responses tailored to users&#x2019; informational, experiential, and emotional needs. To provide human-like emotional support, we use the Willcox Feeling Wheel to guide the chatbot in responding empathetically to the emotions expressed by the user. The chatbot also provides personalized peer experiential expertise support through pragmatic narratives of users in similar situations. We use extensive prompt engineering and fact validation techniques to improve accuracy, personalization, and transparency in generated responses.</p></sec><sec id="s1-3"><title>Protocol Goals</title><p>This protocol describes the development of a RAG chatbot in 2 phases. Phase 1 describes conceptualizing the chatbot architecture, and Phase 2 includes iterative development of the RAG chatbot to provide personalized information, peer experiential expertise, and human-like emotional support. Before prototype development, we conducted a separate study analyzing PrEP-related online social support exchanges to identify the social needs of PrEP candidates to conceptualize the chatbot architecture. Results of this study will be reported in a separate paper. We demonstrate the iterative development process focusing on prompt engineering, which is an integral part of the LLM RAG chatbot implementation. An exploratory user study was conducted to assess the quality of chatbot responses compared to real-world user responses and general LLM responses. We also discuss the implications of these findings, highlighting the feasibility of the chatbot prototype as a tool to facilitate future PrEP support.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>RAG Architecture Overview</title><p>To improve accuracy and empathy, we designed an RAG architecture that explicitly separates informational, experiential expertise, and emotional support. Unlike a traditional LLM, which processes a user query based solely on its internal training data, this RAG framework uses a support classifier to identify specific user needs. The RAG chatbot then queries specialized knowledge bases to deliver tailored, evidence-based responses to the identified social support needs. This modular approach helps in controlling hallucination and improving the accuracy of the support provided. The mechanism involves segmenting user queries into specific needs and retrieving information from separate factual and experiential databases; the system ensures that responses are both evidence-based and empathetically aligned. The key features of the RAG chatbot include the following:</p><list list-type="order"><list-item><p>Modular retrieval: The system first segments and classifies user queries into 3 distinct categories&#x2014;informational, experiential expertise, or emotional support. Using the enriched query segment, the RAG chatbot retrieves documents from specialized databases. It pulls from a clinical facts database for informational needs and an anonymized peer-experience database for experiential and emotional needs. This ensures that the chatbot only references verified data for facts and draws from real-world human narratives to provide empathetic and experiential support.</p></list-item><list-item><p>Contextual enrichment: A dedicated context manager handles query enrichment by ensuring sufficient information is present to respond to the query. If the query is underspecified, the manager prompts the user for additional context to refine the retrieval process.</p></list-item><list-item><p>Response generation: The retrieved context is integrated with specialized system prompts to generate the final response. During this stage, the system leverages a human emotional framework to map the user&#x2019;s expressed feelings to a matching emotional vocabulary. This approach facilitates nuanced emotional support that aligns with the user&#x2019;s specific emotional state and is grounded in real-world human empathy. To maintain cultural authenticity and relatability, the system incorporates direct quotes and real-world narratives from the experience database, providing responses grounded in genuine human experience rather than synthetic advice.</p></list-item></list></sec><sec id="s2-2"><title>Phase I: Prototype Conceptualization</title><p>To inform the goals of our RAG chatbot prototype, we analyzed 3020 publicly available Reddit conversations focused on PrEP from April 2011 to March 2024. We focused on the informational (including peer experiential expertise) and emotional (ie, psychosocial) support because these are the 2 main types of support that are exchanged in this forum and because these have the potential to impact broader behavioral trends related to PrEP use [<xref ref-type="bibr" rid="ref72">72</xref>]. Our database for RAG aligns with the previous literature [<xref ref-type="bibr" rid="ref73">73</xref>]. We found that more than 88% (2658/3020) of user PrEP needs included informational and emotional need items. Informational needs centered on PrEP facts and related experiences, and emotional needs focused on reassurance, HIV and PrEP concerns sharing, empathy, and sympathy, based on the psychosocial constructs of Social Support Behavior Code (SSBC) [<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref75">75</xref>]. PrEP candidates&#x2019; emotions can be described with the Willcox Feeling Wheel [<xref ref-type="bibr" rid="ref69">69</xref>], including sad, mad, scared, joyful, peaceful, and powerful.</p><p>Real-world user queries were multifaceted, consisting of multiple informational and emotional needs. Chatbots must (1) preprocess each input query into semantic segments, (2) generate responses for each segment, and (3) combine them into a final response. Based on literature [<xref ref-type="bibr" rid="ref73">73</xref>] and our prior work, our chatbot was designed to (1) generate personalized responses, (2) provide accurate and comprehensive information to user queries, (3) provide peer experiential expertise support, and (4) respond empathetically to the emotions expressed by the user. Functional dialogue flow diagrams were created (<xref ref-type="fig" rid="figure1">Figure 1</xref>) to visually communicate with our team members and developers how the chatbot would achieve these aims.</p><p>To identify user needs, we created a PrEP-related dataset. This dataset was annotated with 3 categories: informational (including facts and peer experiential expertise), emotional, and context. Subsequently, we built a classifier to parse user inputs and categorize them into the 3 categories. The classifier further classified informational needs into facts and peer experiential expertise. For informational needs, we specifically designed the chatbot to provide information (eg, facts and actionable advice) and peer experiential expertise support (ie, other people&#x2019;s experiences and opinions). For emotional needs, we designed the chatbot to identify user emotions based on the Willcox Feeling Wheel [<xref ref-type="bibr" rid="ref69">69</xref>] and to generate a suitable response using relevant SSBC psychosocial constructs known to impact PrEP uptake (eg, reassurance, understanding, empathy, sympathy, validation, relief of blame, and compliment) [<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref76">76</xref>].</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>An example of a generated response using a paraphrased real-world query from Reddit. The retrieval-augmented generation process includes (1) semantically segmenting the query into sentences and classifying each segment into 1 of the 3 categories: &#x201C;information,&#x201D; &#x201C;context,&#x201D; or &#x201C;emotion&#x201D;; (2) checking for additional context; (3) retrieving the top 5 relevant documents from specific databases; (4) identifying user emotions from emotional segments and retrieving documents via topic matching; and (5) generating a response using personalized prompts. The dotted line in the diagram indicates ongoing updates. LLM: large language model; PrEP: preexposure prophylaxis.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e79966_fig01.png"/></fig></sec><sec id="s2-3"><title>Phase II: Iterative Prototype Development</title><sec id="s2-3-1"><title>Overview</title><p>The iterative development cycle (<xref ref-type="fig" rid="figure2">Figure 2</xref>) consists of four processes: (1) system architecture design&#x2014;drafting initial prompts for guiding LLM response generation, (2) internal evaluation of chatbot responses, (3) modification of the RAG chatbot based on internal evaluation feedback, (4) exploratory user evaluation to compare the RAG chatbot responses against general LLM and real-world user responses, and (5) a heuristic evaluation by HIV experts and a large-scale user study to validate accuracy, safety, ethical alignment, and user trust of chatbot responses prior to real-world deployment.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Iterative development and refinement process of the retrieval-augmented generation chatbot for providing support to preexposure prophylaxis candidates. The process initiated with (1) system architecture design, followed by (2) internal evaluation involving developers and an HIV expert to assess functional requirements. This resulted in a recursive cycle of (3) prototype modification, in which the retrieval-augmented generation chatbot is modified based on internal evaluation and findings from (4) exploratory user study and future (5) heuristic evaluation and large-scale user study. LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e79966_fig02.png"/></fig></sec><sec id="s2-3-2"><title>System Architecture</title><sec id="s2-3-2-1"><title>Overview</title><p>The RAG chatbot was developed using Python programming and by accessing the Google Gemini API. The chatbot consists of 2 main modules (<xref ref-type="fig" rid="figure3">Figure 3</xref>): a preprocessor and a RAG module. The RAG, in turn, leverages 2 databases: a facts database and an experience database. The facts database contains verified information from factsheets, brochures, and websites on HIV and PrEP from public health agencies and private health organizations. The experience database contains Reddit user experiences and emotions related to PrEP concerns. The preprocessor extracts subqueries and context information from user input. To enhance model determinism and minimize irrelevant generation, the information retriever uses the extracted information to handle context management and information subclassification to enhance semantic retrieval and output relevant documents. Finally, the LLM generator synthesizes the retrieved documents and user context through personalized prompt engineering, providing a robust framework for aligning model behavior with user-specific needs while maintaining high factual fidelity.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Architecture of the retrieval-augmented generation chatbot used to generate personalized information, peer experiential expertise, and human-like emotional responses for providing tailored responses. LLM: large language model; PrEP: preexposure prophylaxis; RAG: retrieval-augmented generation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e79966_fig03.png"/></fig></sec><sec id="s2-3-2-2"><title>Preprocessor Module</title><p>The preprocessor module consists of 2 components: a query segmenter and a support classifier.</p><sec id="s2-3-2-2-1"><title>Query Segmenter</title><p>The query segmenter semantically segments a user input into sentences using the SAT tool [<xref ref-type="bibr" rid="ref77">77</xref>]. The tool is highly robust because it is minimally reliant on punctuation and space characters. This makes it an optimal choice for handling diverse writing styles and complex real-world user queries. Each segment is then passed to the support classifier.</p></sec><sec id="s2-3-2-2-2"><title>Support Classifier</title><sec id="s2-3-2-2-2-1"><title>Overview</title><p>The support classifier was used to classify the intent of the input query segments to guide the retrieval process. For segments identified as informational, a few-shot information subclassifier served as a routing layer: it directed experience-based queries to the experiential database and fact-based queries to the factual database. This classification-driven routing ensures that the retrieval pipeline is grounded in the appropriate knowledge database. A detailed description of this routing logic and how classifier labels trigger specific database retrieval is provided in the &#x201C;Information Subcategorization&#x201D; section.</p></sec><sec id="s2-3-2-2-2-2"><title>Dataset</title><p>We created the PrEP dataset, which consists of semantic sentences extracted from online user PrEP discussions and manually labeled them into 1 of the 3 categories&#x2014;informational, emotional, and context. For fine-tuning, we used 2 batches of data: the entire (n=32,081) PrEP dataset (class-unbalanced) and a class-balanced subset (n=10,041) of the PrEP dataset.</p></sec><sec id="s2-3-2-2-2-3"><title>Classifier Fine-Tuning</title><p>We developed 6 support classifiers by fine-tuning each batch of data using 3 models: Bidirectional Encoder Representations from Transformers (BERT)-base-uncased model, Gemma 2B-it [<xref ref-type="bibr" rid="ref78">78</xref>], and Gemma 2 2B-it [<xref ref-type="bibr" rid="ref78">78</xref>]. Gemma model series is lightweight and hardware-flexible, offering low latency and balancing performance and efficiency, making it suitable for chatbot deployments [<xref ref-type="bibr" rid="ref78">78</xref>]. BERT is the state-of-the-art model used in text classification tasks.</p><p>We used Parameter-Efficient Fine-Tuning (PEFT) with standard Low-Rank Adaptation (LoRA) configurations (&#x03B1;=32 and rank=64) [<xref ref-type="bibr" rid="ref79">79</xref>] on a sequential classification task for fine-tuning the Gemma models. Based on previous literature on LLM fine-tuning performance, we fine-tuned the Gemma 2B-it model for 5 epochs with an early stopping patience of 2 epochs [<xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref81">81</xref>]. Similarly, the BERT model was fine-tuned for 5 epochs.</p></sec><sec id="s2-3-2-2-2-4"><title>Evaluation Metrics</title><p>We used <italic>F</italic><sub>1</sub>-score and accuracy to compare the performance of the 6 classifiers on the class-balanced (macro average) and class-unbalanced (weighted average) datasets.</p></sec></sec></sec></sec></sec><sec id="s2-4"><title>RAG Module</title><p>The RAG module consists of an information retriever and an LLM generator.</p><sec id="s2-4-1"><title>Information Retriever</title><p>The information retrieval process includes context management, information subcategorization, and retrieval selection.</p><sec id="s2-4-1-1"><title>Context Management</title><p>The context handler instructs the LLM (Gemini-2.0-flash) using the context handling prompt 1 (CN-1) to determine whether queries require additional context. If no additional context is needed, only the user query is used for retrieving relevant documents. If additional context is needed, our chatbot will ask users for additional context using the context handling prompt 2 (CN-2). With additional context provided by the user, the chatbot then retrieves relevant documents using both the context and user query.</p></sec><sec id="s2-4-1-2"><title>Information Subcategorization</title><p>To generate personalized responses, the information needs were further subcategorized into (1) seeking facts or (2) requests for peer experiential expertise, using the information subclassification few-shot prompt (IC; Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The identified information subcategory (ie, facts or peer experiential expertise) was used to guide the retrieval process. For queries seeking facts, documents were retrieved from the facts database, and for queries seeking peer experiential expertise, documents were retrieved from the experience database. We included the &#x201C;advice&#x201D; subclassification in the few-shot examples as ongoing updates include integrating the &#x201C;MedHelp&#x201D; database, an online health discussion board containing expert advice related to PrEP.</p></sec><sec id="s2-4-1-3"><title>Retrieval Selection</title><p>The retriever uses Sentence-Bidirectional Encoder Representations from Transformers (SBERT) embeddings [<xref ref-type="bibr" rid="ref82">82</xref>] to retrieve relevant documents from the database. We used SBERT embeddings due to their bidirectional contextual understanding [<xref ref-type="bibr" rid="ref82">82</xref>], computational efficiency [<xref ref-type="bibr" rid="ref82">82</xref>], and superior performance in identifying semantically similar sentences, coupled with synonyms and negation lexical variations [<xref ref-type="bibr" rid="ref83">83</xref>], which are better suited to cover various user writing styles. The retriever uses cosine similarity, and the LLM identifies topics for information retrieval; this was done to provide facts for informational or emotional support. We used topic matching to enhance the retrieval of a wide variety of relevant documents from the facts database, including context-rich queries that challenge semantic retrieval [<xref ref-type="bibr" rid="ref84">84</xref>] in short query databases. We manually reviewed the facts database to identify 47 broad HIV-PrEP&#x2013;related topics (<xref ref-type="other" rid="box1">Textbox 1</xref>). To annotate the facts database, we instructed the LLM (Gemini-1.5-flash) to assign the most suitable topic to each query from this list. If no relevant topic is identified, we instructed the LLM to generate a closely relevant topic to have the ability to adapt to unforeseen user needs. We then manually reviewed the annotated topics to include any LLM-generated new topics in the topic list and to merge semantic duplicates. We used Gemini-1.5-flash, the latest model at that time, and retained the topic lists because we manually verified all the LLM-generated topics and were satisfied with the results.</p><boxed-text id="box1"><title> List of topics used in retrieval-augmented generation topic document retrieval for factual support. The HIV-PrEP (pre-exposure prophylaxis) topic list was created by manually reviewing the facts database and updated using large language model topic assignment. Topic document retrieval from the facts database was used to improve sensitivity in unstructured queries.</title><p>(1) PrEP; (2) PrEP eligibility; (3) PrEP effectiveness; (4) PrEP access and cost in Australia; (5) PrEP access and cost; (6) HIV testing, (7) HIV self-test; (8) postexposure prophylaxis (PEP); (9) PrEP administration and dosage; (10) apreptude or injectable PrEP; (11) PrEP side effects; (12) HIV and AIDS; (13) risk and transmission; (14) HIV prevention strategies; (15) safety and precautions; (16) cabotegravir; (17) sexually transmitted infections (STIs); (18) viral load or undetectable or untransmissible; (19) HIV treatment; (20) PrEP medications; (21) emtricitabine or tenofovir disoproxil fumarate; (22) emtricitabine or tenofovir disoproxil fumarate administration; (23) emtricitabine or tenofovir disoproxil fumarate side effects; (24) drug interactions; (25) emtricitabine or tenofovir disoproxil fumarate eligibility; (26) antiretroviral or HIV drug side effects; (27) test disclosure and anonymity; (28) stigma and its impact on people living with HIV; (29) PEP access and cost; (30) PEP side effects; (31) Apretude or injectable PrEP side effects; (32) cabotegravir side effects; (33) PEP eligibility; (34) PrEP access and cost in the United Kingdom; (35) medication adherence; (36) Truvada side effects; (37) healthy lifestyle; (38) Descovy; (39) PrEP safety and precautions; (40) HIV prevention strategies; (41) Truvada; (42) PrEP eligibility in the United Kingdom; (43) Descovy side effects; (44) HIV or AIDS drugs; (45) Truvada access and cost; (46) Descovy access and cost; and (47) emtricitabine or tenofovir disoproxil fumarate safety and precautions.</p></boxed-text></sec><sec id="s2-4-1-4"><title>Retrieval for Factual Support</title><p>For factual support, we retrieved the top 5 documents from the facts database with a cosine similarity score &#x2265;0.65. The cosine similarity thresholds were determined based on experiments, and we set k=5 for document retrieval based on previous evaluation benchmarks [<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref86">86</xref>]. In addition, we selected 5 topically relevant documents, using the topic identified from the 46 HIV-PrEP topics (<xref ref-type="other" rid="box1">Textbox 1</xref>). To identify the topic of an informational query, we instructed the LLM (Gemini-2.0-flash) using the HIV-PrEP topic list and the personalized information prompt-1 (PI-1) to assign one suitable topic to the user query. Retrieving documents using cosine similarity was based on the context need. If context is needed, we retrieved 3 sets of top 5 relevant documents (<xref ref-type="fig" rid="figure4">Figure 4</xref>) by comparing cosine similarity between (1) the context-added query and database questions (m1), (2) the context-added query and database answers (m2), and (3) the context-free query and database questions (m3; <xref ref-type="fig" rid="figure4">Figure 4</xref>). Both database questions and answers were semantically searched to improve the retrieval precision. Cosine similarity between the context-free query and database answers did not improve the retrieval precision and was therefore not used in processing the context-free queries. As the majority of the facts database consists of short questions from PrEP factsheets and brochures, the retriever often failed to retrieve relevant information for queries with additional context. If no context is needed, we retrieved the top 5 documents by only comparing cosine similarity between the context-free query and the database questions.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Retrieval algorithm pseudocode for improving retrieval precision in factual queries using cosine similarity.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e79966_fig04.png"/></fig></sec><sec id="s2-4-1-5"><title>Retrieval for Peer Experiential Expertise and Emotional Support</title><p>For peer experiential expertise support, we retrieved the top 5 documents from the experience database using cosine similarity with a score of at least 0.65. When context is needed, we separately used both the query and context for retrieval to overcome context-dependent retrieval failure. For emotional queries, the top 5 similar documents were retrieved from the experience database in addition to providing relevant facts. Similar to document retrieval for peer experiential expertise support, we retrieved the top 5 documents from the experience database using cosine similarity with a score of at least 0.65 and, when context is needed, used both the query and context for retrieval. As information can be helpful in addressing certain HIV-PrEP emotional concerns [<xref ref-type="bibr" rid="ref87">87</xref>], we retrieved facts relevant to the emotional query.</p><p>To improve the sensitivity, we also used up to 3 paraphrase queries with a lower cosine similarity score, between 0.6 and 0.65. For paraphrasing, we used the T5 paraphrase generator from the Transformers Hugging Face library [<xref ref-type="bibr" rid="ref88">88</xref>] for its demonstrated superior performance over GPT-3 and ChatGPT in capturing semantic relevance between sentences [<xref ref-type="bibr" rid="ref89">89</xref>].</p></sec></sec></sec><sec id="s2-5"><title>LLM Generator</title><p>Once all the necessary data were retrieved, we instructed the LLM (Gemini-2.0-flash) to generate responses using the retrieved information. This process was guided by personalized prompts (PI [personalized information], PE [personalized emotion], and PX [personalized experience]) and the user need (ie, informational, peer experiential expertise, and emotional responses) as described in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. We iteratively refined the prompts via prompt engineering to improve comprehensiveness, accuracy, relevancy, transparency, and response coherence. The PI-2 prompt instructs the LLM to provide a detailed response, followed by the PI-3 prompt to check the consistency of factual queries. For peer experiential expertise queries, we used the PX prompt that instructs the LLM to use the retrieved documents and generate a third-person narrative with explicit statements of partner conversations and direct quotes, when possible, to emphasize transparency.</p><p>For providing human-like emotional support, we used PE-1 to first identify the feelings based on the 6 Willcox feeling categories and to generate a suitable response reflecting SSBC psychosocial constructs (ie, sympathy, understanding, empathy, relief of blame, validation, compliment, or reassurance) for the identified feeling. Identifying user feelings allows responding with suitable emotional expressions that are typically associated with human responses. We also found that general LLM responses to emotional queries lacked deeper understanding of underlying emotions and human-like emotional behaviors, consistent with findings in previous literature [<xref ref-type="bibr" rid="ref90">90</xref>]. The RAG LLM was instructed to use the style and tone from the retrieved emotional documents, which comprise peer-to-peer emotional support. When no relevant documents were retrieved for an emotional query, we used the PE-2 to generate a suitable response reflecting SSBC psychosocial constructs for the identified user feeling, without using the style and tone from the retrieved documents. Fact responses were validated by the fact validator. Then, the cohesive response (CR) prompt combines all subquery responses, instructing the LLM to smooth the information flow into a single CR. Finally, the LLM provides a combined response.</p></sec><sec id="s2-6"><title>Fact Validator</title><p>Responses providing factual information were validated with the documents from the facts database, a step toward eliminating LLM hallucination. Our fact validation algorithm consists of 2 steps: fact identification and fact validation. To reduce overreliance on LLMs, we used a deterministic technique to identify semantically relevant facts from the retrieved documents (ie, the same top 5 documents retrieved from the information retriever process). This is done to address a known issue: that LLMs often provide imperfect fact identification when they break down complex information [<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]. To identify the most relevant facts for validation, we (1) segmented the generated response and retrieved documents using the SAT tool and then (2) extracted the most relevant fact-supporting document segment (step 1) based on SBERT and cosine similarity (ie, highest similarity score). This creates pairs of generated response segments that correspond to their fact-supporting document segments. The LLM was instructed to use the response-document pairs and the FV-1 prompt (step 2) to identify any response segments that are not supported by the document segments identified with cosine similarity. Providing the subset of relevant fact segments to the LLM narrows its focus, improving validation accuracy. Then, using any generated response segments with inaccurate facts and the FV-2 prompt, the LLM was instructed to extract supporting facts from the retrieved documents (step 3).</p><p>To evaluate the performance of the fact validator, we manually verified the performance for 1033 fact response segments from 100 generated responses for randomly sampled query responses that the LLM generated. The fact validator achieved a 97.39% validation accuracy based on 1006 sample items. The accuracy was calculated as the percentage of fact response segments that were fully supported by the facts database, as determined by the research team. Only 3% (27/1033) of the generated response segments were not able to be validated with the facts database (95% CI 1.85&#x2010;3.88). These included response segments providing general information (n=12, eg, <italic>&#x201C;</italic>Other side effects [of the medication] include changes in the immune system, dizziness, diarrhea, nausea, vomiting, headache, rash, and gas<italic>.</italic>&#x201D;) and personalized information (n=8, eg, &#x201C;Given the recent unprotected anal sex with someone who recently learned they had syphilis, your friend should get tested for STIs as soon as possible.&#x201D;). We also found false negatives (n=7), where the LLM extracted facts that could not support the generated response segment in step 3. For example, when validating a response segment (eg, &#x201C;Truvada alone is not sufficient for post-exposure prophylaxis (PEP).&#x201D;) with an accurate and semantically retrieved document segment<italic>,</italic> the LLM extracted facts that could support the generated response segment (eg, &#x201C;It is worth noting that Truvada must be used correctly (without skipping doses) whether it is used in combination with other ARVs in the control of HIV, or for PrEP.&#x201D;). Based on these data, we opted for a 3-step validation process because the semantic retrieval (step 1) identified over 82.10% (848/1033) of accurate response segments, whereas the LLM identified (step 2 and step 3) 15.30% (158/1033) of accurate response segments. It is also important to note that we also found that LLM-based validation was less consistent compared to the deterministic techniques. For instance, when we processed the same sets of 100 samples 10 times, the LLM changed its final validation responses (step 2 and step 3) in 44% of samples. It is important to note that the 44% does not indicate that earlier responses were inaccurate. Rather, the nondeterministic nature of LLMs (as probabilistic models) relying solely on LLMs can hinder the validation process. This highlights the need for mixed validation approaches, combining deterministic techniques and LLMs to eliminate LLM hallucinations.</p></sec><sec id="s2-7"><title>Internal Evaluation of the Chatbot Responses and Prompt Refinement</title><sec id="s2-7-1"><title>Internal Evaluation</title><p>The RAG chatbot was evaluated to assess the functional requirements following the iterative evaluation design process [<xref ref-type="bibr" rid="ref93">93</xref>], with a focus on (1) accurately identifying user informational and emotional needs, (2) providing relevant and accurate information, and (3) generating personalized and emotionally suited responses. Developers conducted 10 rounds of internal evaluation of chatbot responses. Paraphrased user inputs from the experience database were used to create 2 test queries for each support need (ie, informational, peer experiential expertise, and emotional). In each of the 10 evaluation rounds, we selected (1) one simple and direct query and (2) one complex and context-dependent query for each support need. Responses were generated for the 6 test queries per round, which were then iteratively reviewed in 10 rounds of internal evaluation (n=60) by at least 2 chatbot developers and at least 1 HIV expert. This qualitative evaluation of 60 total responses helped to identify inaccuracies and refine the LLM prompts and was followed by chatbot modification to address issues identified by the reviewers. The most frequently occurring issues were as follows: (1) low retrieval sensitivity and precision for context-dependent queries, (2) contradicting information from user discussions, (3) response generation using inherent knowledge, (4) irrelevant response context, (5) chatbot fictional narratives, (6) incomprehensive responses, and (7) noisy text (eg, unwanted phrases and special characters). Generated responses were evaluated based on the following 10 criteria: clarity [<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref95">95</xref>], accuracy [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref96">96</xref>], actionability [<xref ref-type="bibr" rid="ref96">96</xref>], relevancy [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>], information detail [<xref ref-type="bibr" rid="ref95">95</xref>], tailored information [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>], comprehensiveness (questions answered fully) [<xref ref-type="bibr" rid="ref95">95</xref>], language suitability [<xref ref-type="bibr" rid="ref94">94</xref>], tone [<xref ref-type="bibr" rid="ref95">95</xref>], and empathy [<xref ref-type="bibr" rid="ref89">89</xref>]. The responses were also manually reviewed to verify whether the chatbot generates responses only using inherent knowledge (ie, verifying whether generated content is outside the scope of retrieved documents).</p></sec><sec id="s2-7-2"><title>Prompt Engineering</title><sec id="s2-7-2-1"><title>Overview</title><p>First, we created level-1 prompts (ie, PI, PX, PE, and CR) that were simple and direct instructions for generating a suitable response to a user query [<xref ref-type="bibr" rid="ref97">97</xref>]. The responses generated using level-1 prompts were then internally evaluated to assess functional requirements. Initially, the LLM rarely generated inaccurate information based on the user experience database. After we finalized the RAG database architecture, we followed LLM prompt engineering techniques [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref97">97</xref>]. Prompts were iteratively refined following 10 rounds of internal evaluation to result in the final level-2 prompts (ie, PI, PX, PE, and CR) that were more specific and structured to enhance information comprehensiveness, accuracy, relevancy, transparency, and response coherence.</p><p>All prompts were checked sequentially for overlapping effects. The process started with information subclassification (IC) and context handling prompts (CN-1 and CN-2). Then, we implemented personalized information prompts, followed by prompts for topic identification (PI-1), personalized information response generation (PI-2), and verifying relevancy (PI-3). Then, the PX and personalized emotion prompts (PE-1 and PE-2) were developed. Finally, we developed and refined the prompt for response cohesion (CR). The final prompts are shown in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-7-2-2"><title>Prompt Engineering for Information Comprehensiveness</title><p>To improve information comprehensiveness and low retrieval sensitivity and precision, we refined the level-1 PI-2 prompt by instructing the LLM to segment the user query and provide a detailed response to all segments using the retrieved documents. In contrast, our initial level-1 PI-2 prompt (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) simply instructed the LLM to provide a response to a user query using the retrieved documents; this generated responses that were not comprehensive (ie, the question was not answered fully).</p></sec><sec id="s2-7-2-3"><title>Prompts for Response Accuracy</title><p>We refined our level-1 IC, CN-1, PI-2, PE-1, FV-1, FV-2, and PX prompts (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) to improve the accuracy of generated responses. These prompt refinements helped address issues related to low retrieval sensitivity and precision, contradictory information, response generation using inherent knowledge, and fictional narratives. We refined the zero-shot IC prompt to a few-shot IC prompt to improve the accuracy of subclassifying informational needs. To improve context handling, the level-1 CN-1 prompt was iteratively paraphrased to verify whether additional context is needed for queries.</p><p>To enhance the accuracy of generated information, we refined the PI-2, FV-1, FV-2, and PX prompts to emphasize (by using &#x201C;double asterisk&#x201D; in the LLM prompts) the use of only the retrieved documents for generating a response. Using the level-1 PI-2 prompt, the LLM still relied on inherent knowledge in responding to factual queries, even when instructed to use the retrieved factual documents to generate a response. The level-1 FV-1 and FV-2 prompts failed to identify facts that supported generated response segments. This was primarily due to a lack of clarity in defining factual consistency. This led to false positives, in which the LLM identified facts that only partially support the response segments. To prevent generating fictional real-world experiences, we refined the level-1 PX prompt, which instructed the LLM to clearly indicate other people&#x2019;s experiences or opinions as a third-person narrative using the documents retrieved from the experience database.</p><p>To avoid inaccurate but anecdotal information in the experience database, we refined the PE-1 prompt to instruct the LLM to avoid using facts retrieved from the experience database to provide factual support. Although the level-1 PE-1 prompt included an instruction to use the retrieved documents only to reflect style and emotional tone, the LLM still generated factual support using the experience database, which was identified when a response contained some inaccurate opinions retrieved from the experience database. For example, the experience database contained a user post &#x201C;...The risk, even if he is undetectable, isn't zero, but it&#x2019;s pretty low...,&#x201D; which contradicts the recognized medical information [<xref ref-type="bibr" rid="ref98">98</xref>].</p></sec><sec id="s2-7-2-4"><title>Prompts for Information Relevance</title><p>To strengthen information relevance, we developed the PI-3 prompt (refer to Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> for the final refined prompt) that verified the consistency of generated responses with the user query. The level-1 PI-2 prompt generated responses that lacked contextual awareness, specifically regarding locations or populations. Refining the PI-2 prompt to have the additional task of verifying response consistency proved ineffective. This was a consistent theme in prompt engineering. To reduce the complexity of prompting, we developed a separate PI-3 prompt to verify consistency after generating a response using the PI-2 prompt.</p></sec><sec id="s2-7-2-5"><title>Prompts for Information Transparency</title><p>To prevent generation of fictional narratives, we refined the PX prompt by instructing the LLM to include direct quotes of the user&#x2019;s personal experiences when available and requiring the LLM to explicitly state that the response information was based on other people&#x2019;s experiences (eg, &#x201C;... This response is based on other people&#x2019;s experience.&#x201D;).</p></sec><sec id="s2-7-2-6"><title>Prompts for Response Coherence</title><p>To improve readability, comprehension, and user experience, we refined the PI-1, PI-2, PE-1, and PE-2 prompts to remove noisy responses, specifically unnecessary special characters (eg, * in &#x201C;**Cost of PrEP in the US:**...&#x201D; and [ in &#x201C;[PrEP provided...infection.]&#x201D;) and logical reasoning, which included chatbot-provided explanations for prompt execution (eg, &#x201C;Okay, here are the revised responses, aiming for smoother flow and a more natural conversational tone: ...&#x201D;) in responses.</p></sec></sec></sec><sec id="s2-8"><title>Exploratory User Evaluation</title><sec id="s2-8-1"><title>Study Design and Recruitment</title><p>An exploratory blinded-comparative online user study was conducted to evaluate the RAG chatbot&#x2019;s output performance, where participants were presented with 3 sets of responses for each query: a real-world user response, a RAG-generated response, and a general LLM baseline (Gemini-2.0-flash). This study serves as an exploratory content validation and protocol design phase, establishing the foundational functional readiness of the system as well as future user studies. A total of 7 participants were recruited through personal networks and students at the authors&#x2019; institution.</p></sec><sec id="s2-8-2"><title>Data Sampling</title><p>A batch of 25 queries and corresponding user responses were randomly sampled from the experience dataset (Reddit). For each query, three response types were compared: (1) user response: real-world user response from Reddit, (2) RAG chatbot response: responses generated by the proposed RAG chatbot, and (3) general LLM: responses from the Gemini-2.0-flash model (temp=0.65).</p><p>The evaluation was delivered via structured online questionnaires (Google Forms). Each participant evaluated a randomly selected subset of 15 queries. To ensure internal validity, the 3 response types were shuffled and deidentified for every query, ensuring participants rated the content quality without knowledge of the response source.</p></sec></sec><sec id="s2-9"><title>Measures</title><p>Two types of questionnaires were developed for the study using Google Forms. The demographic questionnaire collected participant information on demographic and professional experience. The questionnaire consists of 6 deidentified questions asking participants about their age, gender, ethnicity, race, region, and professional background.</p><p>The chatbot response evaluation questionnaire was used to assess the quality of the responses based on 10 criteria (<xref ref-type="table" rid="table1">Table 1</xref>): information accuracy [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref96">96</xref>], information clarity [<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref95">95</xref>], information actionability [<xref ref-type="bibr" rid="ref96">96</xref>], information detail [<xref ref-type="bibr" rid="ref95">95</xref>], relevance [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>], tailored information [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>], comprehensiveness-question answered fully [<xref ref-type="bibr" rid="ref95">95</xref>], suitable tone [<xref ref-type="bibr" rid="ref95">95</xref>], empathy [<xref ref-type="bibr" rid="ref89">89</xref>], and suitable language [<xref ref-type="bibr" rid="ref94">94</xref>]. The questionnaire contained instructions and definitions of each criterion along with the set of PrEP queries and corresponding responses. Participants were required to rate each response on a 5-point Likert scale for all criteria. Based on prior literature on expert-reviewed chatbot responses [<xref ref-type="bibr" rid="ref99">99</xref>] suggesting good to excellent performance, we determined a response quality rating of above 4 on a 5-point scale for each of the 10 evaluation criteria.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Evaluation criteria and descriptions provided in the chatbot response evaluation questionnaire.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Attribute and criterion</td><td align="left" valign="bottom">Description</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Information quality</td></tr><tr><td align="left" valign="top">Information accuracy [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref96">96</xref>]</td><td align="left" valign="top">Is the information provided factually correct and free from misinformation?</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Information clarity [<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref95">95</xref>]</td><td align="left" valign="top">Is the information presented in a clear, concise, and easy-to-understand manner?</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Information actionability [<xref ref-type="bibr" rid="ref96">96</xref>]</td><td align="left" valign="top">Does the response provide practical, actionable advice that the user can implement?</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Information detail [<xref ref-type="bibr" rid="ref95">95</xref>]</td><td align="left" valign="top">Does the response provide sufficient detail?</td></tr><tr><td align="left" valign="top" colspan="2">Response relevance and appropriateness</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Relevance [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>]</td><td align="left" valign="top">Does the response directly address the user&#x2019;s question and concerns?</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Tailored information [<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref94">94</xref>]</td><td align="left" valign="top">Is the information provided tailored to the specific needs and circumstances of the user?</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Comprehensiveness: question answered fully [<xref ref-type="bibr" rid="ref95">95</xref>]</td><td align="left" valign="top">Does the response fully answer the question without leaving significant gaps or requiring further clarification?</td></tr><tr><td align="left" valign="top" colspan="2">Language and tone</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Suitable tone [<xref ref-type="bibr" rid="ref95">95</xref>]</td><td align="left" valign="top">Is the tone of the response appropriate for the context and the user&#x2019;s overall emotional state?</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Respectful, empathetic, and considerate manner [<xref ref-type="bibr" rid="ref89">89</xref>]</td><td align="left" valign="top">Does the response demonstrate empathy and understanding for the user&#x2019;s situation? Is the response respectful and non-judgmental?</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Suitable language [<xref ref-type="bibr" rid="ref94">94</xref>]</td><td align="left" valign="top">Is the language free from jargon or technical terms that may be unfamiliar to the user?</td></tr></tbody></table></table-wrap></sec><sec id="s2-10"><title>Statistical Analysis</title><p>To account for the nonindependence of observations where multiple ratings were provided by the same participants, a linear mixed-effects model was used using the <italic>lme4</italic> package in R (v4.3.1; R Core Team). With 7 participants evaluating 15 queries across 3 response types and 10 evaluation criteria, the model analyzed a total of 3150 response evaluations. Participants and queries were treated as random effects to control individual rater bias and item variability. Quantitative analysis was conducted using R (R Core Team) and Microsoft Excel.</p></sec><sec id="s2-11"><title>Phase III: Planned Prototype Enhancement and Heuristic Evaluation</title><sec id="s2-11-1"><title>Future Prototype Enhancements</title><p>To further improve the quality and diversity of data, we will expand the data sources to include 2 additional discussion boards from MedHelp and POZ community (HIV positive) to broaden the knowledge base. MedHelp and POZ databases consist of PrEP-related conversational dialogues among community members. We will develop prompts specifically tailored for varying formats and structures [<xref ref-type="bibr" rid="ref100">100</xref>], including conversational dialogues, question-answer pairs, and dialogues with quotes. We will update the chatbot architecture to maintain historical dialogues to reflect real-world multiturn conversations and include denial capabilities to set clear expectations for users and ensure safety compliance. We will also improve the fact validator&#x2019;s accuracy to reduce the impact of LLM&#x2019;s inherent inconsistencies when extracting facts to support generated responses. The final verification will focus on the accuracy of the responses rather than on the fact validator&#x2019;s performance using the retrieved documents.</p></sec><sec id="s2-11-2"><title>Planned Heuristic Evaluation: Expert Evaluation</title><p>To assess chatbot response quality, an iterative heuristic evaluation will be conducted with 5&#x2010;8 HIV experts. Eligible participants will have professional experience as an HIV counselor, researcher, health practitioner (eg, HIV care providers), pharmacist, and/or social and community health worker. After providing online consent, participants will complete an online demographic questionnaire followed by the chatbot response evaluation questionnaire. The demographic questionnaire consists of 6 deidentified questions asking participants about their age, gender, ethnicity, race, region, professional title, years of experience as an HIV expert, and frequency of interaction with patients. The evaluation questionnaire will use a 5-point Likert scale to assess 10 evaluation criteria (<xref ref-type="table" rid="table1">Table 1</xref>) for 6 pairs of user queries and chatbot responses. Participants will have the opportunity to provide some additional information and human sample responses to help improve chatbot responses.</p><p>To measure the consistency of ratings provided by the experts, we will calculate the intraclass correlation coefficient (ICC) [<xref ref-type="bibr" rid="ref101">101</xref>] for the ratings of each criterion. We will then compare the chatbot-generated responses to expert responses by measuring Bilingual Evaluation Understudy (BLEU; IBM researchers Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu) [<xref ref-type="bibr" rid="ref89">89</xref>] and Recall-Oriented Understudy for Gisting Evaluation - Longest Common Subsequence (ROUGE-L; Chin-Yew Lin) [<xref ref-type="bibr" rid="ref102">102</xref>] scores. These scores computationally measure relevance and linguistic similarity, and are scalable for evaluating larger datasets. In addition, human emotions are complex and are influenced by individual perception; this makes generating text for personalized emotional support a highly challenging endeavor.</p><p>Results and analysis of expert opinions will be published in a subsequent paper.</p></sec><sec id="s2-11-3"><title>Future Large-Scale User Evaluation</title><p>To evaluate the system&#x2019;s real-world utility, we will conduct a large-scale user study involving 50 PrEP candidates. Participants will be recruited in coordination with HIV prevention experts and established PrEP programs, using public resources such as referral lists, health websites, and online directories. All study procedures will be conducted in accordance with institutional review board (IRB) standards, with recruitment facilitated through IRB-approved flyers and digital outreach materials.</p><p>Eligibility criteria include that participants must be at least 18 years of age, proficient in English, and meet clinical eligibility for PrEP. Following the provision of informed online consent, participants will complete a series of comprehensive study questionnaires followed by a concluding interview. The estimated time commitment for the session is between 90 and 120 minutes.</p><p>To ensure methodological consistency across evaluation phases, this user study will use the same 10-point evaluation criteria and questionnaires used in the exploratory user evaluation. By incorporating qualitative measures (eg, semistructured interviews), we aim to gain deeper insights into user experience, perceived trust, and the effectiveness of the chatbot-provided PrEP support. This will allow us to assess how the chatbot performs in real-world scenarios beyond research settings.</p></sec><sec id="s2-11-4"><title>Ethical Considerations</title><sec id="s2-11-4-1"><title>Institutional Approval</title><p>The exploratory user evaluation was reviewed and approved by the IRB at the University of North Carolina at Charlotte (protocol number IRB-26&#x2010;0423). The study was determined to pose minimal risk to participants, as the data collection was limited to subjective evaluations of chatbot responses and did not collect any personally identifiable information. For the planned heuristic evaluation and subsequent large-scale user study, separate IRB applications will be submitted for approval prior to recruitment, ensuring all ethical standards are met for that specific population.</p></sec><sec id="s2-11-4-2"><title>Participant Consent and Incentives</title><p>Informed consent is obtained electronically via DocuSign (DocuSign Inc Team) from all participants, including students, community members, and HIV experts, prior to study commencement. Participants are informed about the study details, potential risks, participation rights, and incentives. They are explicitly informed that their participation was voluntary and that they could withdraw at any time.</p><p>In the exploratory user evaluation, students received the opportunity to replace one class lab activity as an incentive, while other participants received no compensation; an alternative nonresearch activity was available to students to ensure the voluntary nature of participation. Participants in the expert and large-scale user study will receive a US $25 Amazon gift card on completion of the study, subject to IRB approval.</p></sec><sec id="s2-11-4-3"><title>Privacy and Data Security</title><p>To protect participant privacy, all data were or will be collected and analyzed anonymously, ensuring that no feedback could be traced back to individual experts or professionals. The dataset is filtered to pseudonymize any PII information such as names, email addresses, contact information, and IP addresses during the RAG process.</p></sec></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>We have developed a RAG chatbot prototype as of January 2025 and have iteratively refined the chatbot design and system based on feedback from internal evaluation.</p></sec><sec id="s3-2"><title>Support Classifier Performance</title><p>To assess the reliability of the RAG chatbot in identifying support needs, we validated the 3-class support classifier in categorizing queries into informational, emotional, and contextual needs.</p><p>As reported in <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>, our fine-tuned Gemma 2 2B-it model demonstrated strong performance in classifying PrEP support needs when fine-tuned on the full dataset (<italic>F</italic><sub>1</sub>-score=0.86) and class-balanced subset (<italic>F</italic><sub>1</sub>-score=0.89). Specifically, we found that fine-tuning the Gemma 2 2B-it model on a class-balanced subset indicated strong accuracy and precision in classifying emotional needs, while maintaining comparable performance in classifying informational and contextual queries (<xref ref-type="table" rid="table2">Table 2</xref>). Based on superior performance indicated by <italic>F</italic><sub>1</sub>-score metrics, we used the class-balanced Gemma 2 2B-it model during preprocessing for classifying user support needs. In the context of classifying informational and emotional support seeking, our class-unbalanced BERT model exhibits superior performance (<italic>F</italic><sub>1</sub>-information score=0.92, <italic>F</italic><sub>1</sub>-emotion score=0.71) compared to a previous social support seeking dataset, CHQ-SocioEmo (<italic>F</italic><sub>1</sub>-information score=0.58, <italic>F</italic><sub>1</sub>-emotion score=0.62). This classification of user queries allows our chatbot to identify user needs, extract context, and generate tailored and emotionally relevant responses.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Results of an assessment of the performance of 3 classifiers, fine-tuned on the class-balanced PrEP<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> dataset. These classifiers are used for classifying user input into 3 classes.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Class</td><td align="left" valign="bottom" colspan="2">BERT-base-uncased<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="bottom" colspan="2">Gemma 2B-it</td><td align="left" valign="bottom" colspan="2">Gemma 2 2B-it</td></tr></thead><tbody><tr><td align="left" valign="top"/><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Accuracy</td></tr><tr><td align="left" valign="top">Information</td><td align="left" valign="top">0.93</td><td align="left" valign="top">0.94</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.93</td><td align="left" valign="top">0.90</td></tr><tr><td align="left" valign="top">Emotion</td><td align="left" valign="top">0.85</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.86</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.87</td><td align="left" valign="top">0.90</td></tr><tr><td align="left" valign="top">Context</td><td align="left" valign="top">0.85</td><td align="left" valign="top">0.82</td><td align="left" valign="top">0.86</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.86</td><td align="left" valign="top">0.86</td></tr><tr><td align="left" valign="top">Macro average</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.88</td><td align="left" valign="top">0.89<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">0.89<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>PrEP: preexposure prophylaxis.</p></fn><fn id="table2fn2"><p><sup>b</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table2fn3"><p><sup>c</sup>Best-performing model.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Results of an assessment of the performance of 3 classifiers fine-tuned on the full PrEP<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> dataset. These classifiers are used for classifying user inputs into 3 classes.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Class</td><td align="left" valign="bottom" colspan="2">BERT-base-uncased<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="bottom" colspan="2">Gemma 2B-it</td><td align="left" valign="bottom" colspan="2">Gemma 2 2B-it</td></tr></thead><tbody><tr><td align="left" valign="top"/><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Accuracy</td></tr><tr><td align="left" valign="top">Information</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.929</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.927</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.92</td></tr><tr><td align="left" valign="top">Emotion<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.693</td><td align="left" valign="top">0.69</td><td align="left" valign="top">0.624</td><td align="left" valign="top">0.71</td><td align="left" valign="top">0.663</td></tr><tr><td align="left" valign="top">Context</td><td align="left" valign="top">0.95</td><td align="left" valign="top">0.948</td><td align="left" valign="top">0.95</td><td align="left" valign="top">0.964</td><td align="left" valign="top">0.96</td><td align="left" valign="top">0.966</td></tr><tr><td align="left" valign="top">Weighted average</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.922</td><td align="left" valign="top">0.92</td><td align="left" valign="top">0.93<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>PrEP: preexposure prophylaxis.</p></fn><fn id="table3fn2"><p><sup>b</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table3fn3"><p><sup>c</sup>Poor performance compared to class-balanced fine-tuned model predictions.</p></fn><fn id="table3fn4"><p><sup>d</sup>Model with best performance.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Prompt Engineering</title><p>We manually observed that prompt engineering plays a significant role in harnessing LLM capabilities to generate desired outputs. We report key findings from our iterative prompt engineering, offering suggestions for effective prompting in chatbot applications. Although there is no standard approach for refining prompts [<xref ref-type="bibr" rid="ref103">103</xref>], starting with simple and direct instructions (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) helped in understanding LLM comprehension and reasoning ability. For complex and longer queries, we found that prompting the LLM to decompose the query into segments helps in improving the response comprehensiveness (eg, PI-2 prompt from <xref ref-type="table" rid="table1">Table 1</xref>). Such prompt decomposition was also effective in responding to subtle user emotions by first identifying the emotions from a list of emotion categories and guiding the LLM in social-emotional reciprocation for personalizing emotional responses (ie, PE-1 and PE-2 prompts).</p><p>An important finding in LLM prompt engineering was the limitation of using multiple instructions in a single prompt. We found that LLMs indicate low prompt adherence when a prompt has multiple instructions. In such cases, the LLM often defaults to the last instruction, resulting in partial and inconsistent execution of complex logical reasoning prompts. Another challenge was the LLM&#x2019;s sensitivity to linguistic variability in prompts [<xref ref-type="bibr" rid="ref104">104</xref>]. We found that paraphrasing prompts by modifying prompt structure, using synonyms, and expanding prompt context is ineffective in context management tasks (eg, identifying the need for additional context), exhibiting inconsistent response output. We found that shorter prompts with content words summarizing the main concept or idea can reduce ambiguity in LLM decisions.</p><p>A few-shot prompting approach is effective in text classification [<xref ref-type="bibr" rid="ref49">49</xref>]. We found that few-shot prompting did not consistently perform well across classification for informational needs, emotional needs, and context. This limitation is potentially because these categories represent high-level support needs that require diverse perspectives for inferring ambiguous context and broad scope. Consequently, fine-tuning LLMs on annotated datasets produced more reliable results than few-shot prompting. Few-shot prompting is effective in simple subclassification tasks with clear and distinct criteria for narrow scope [<xref ref-type="bibr" rid="ref105">105</xref>].</p></sec><sec id="s3-4"><title>Fact Validation</title><p>For fact validation, deterministic techniques, including semantic relevance and text similarity metrics (eg, ROUGE and BLEU), often failed due to their limited ability to capture negations [<xref ref-type="bibr" rid="ref106">106</xref>] and numerical accuracy [<xref ref-type="bibr" rid="ref107">107</xref>]. We found that LLMs generally struggle to validate statements that are generalized or personalized and are not directly present in the knowledge base. When required facts are scattered across documents, such generalized or personalized statements need implicit validation and cross-referencing for accurate verification of information [<xref ref-type="bibr" rid="ref108">108</xref>].</p></sec><sec id="s3-5"><title>Exploratory User Study Analysis</title><p>The exploratory user study included 7 participants who completed the full evaluation protocol; the majority of participants were female (5/7, 71.4%) and aged 25&#x2010;34 years (5/7, 71.4%), with an Asian (5/7, 71.4%) ethnic background. Across 15 real-world queries evaluated against 10 quality criteria, the study generated 3144 valid observations, with 6 &#x201C;NA&#x201D; ratings (instances where a criterion was deemed nonapplicable by the rater).</p></sec><sec id="s3-6"><title>Preliminary Comparison of RAG Responses With General LLM and Real-World User Responses</title><p>The overall performance of the 3 response sets (general LLM, RAG chatbot, and user) was evaluated across 10 distinct evaluation criteria. Due to nonindependence of data points, a linear mixed model was used to account for user bias and random effects from both participants and specific questions.</p><p>Initial observations indicate variations in response quality across the 3 groups, suggesting specific areas for improvement in the RAG chatbot architecture and prompt design based on the evaluation criteria. Both AI modalities (general LLM and RAG) had higher preference across the evaluation criteria compared to user responses. We observed that the general LLM&#x2019;s communication style appeared to influence how participants rated comprehensiveness, detail, and tailoredness of the information received, though both AI models showed similar performance in terms of empathy and tone.</p><p>Given the exploratory nature of this pilot evaluation, the sample size (n=7) was small and skewed in terms of participant demographics (eg, primarily consisting of university students), which limits the generalizability of these findings to the broader PrEP-candidate population. However, these exploratory insights into future modifications demonstrate the viability of the evaluation framework and provide a descriptive baseline for the heuristic and user study outlined in this protocol.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>We describe the development of a RAG chatbot prototype for providing personalized support to fulfill informational, peer experiential expertise, and emotional needs of PrEP candidates to facilitate PrEP care. The chatbot is designed for (1) identifying support needs and personalizing responses (tailored information, clarity, and tone); (2) providing complete, relevant, and accurate HIV and PrEP information (comprehensiveness, detail, relevancy, accuracy, and language); (3) providing peer experiential expertise support (relevancy and actionability); and (4) responding with human-like emotional support (empathy).</p></sec><sec id="s4-2"><title>Performance Gap of AI Modalities</title><p>Our exploratory user study results provide initial empirical support for the proposed chatbot functionality. Both the RAG chatbot and the general LLM were preferred over user responses across all 10 evaluation criteria. This suggests that AI-generated support can offer an enhanced level of perceived utility, informational depth, and structural coherence compared to traditional peer support related to PrEP.</p></sec><sec id="s4-3"><title>Addressing Technical Challenges</title><p>The analysis revealed a performance gap between the two AI architectures. The general LLM performed better than the RAG chatbot in 8 of 10 categories, with the exceptions of &#x201C;empathy&#x201D; and &#x201C;tone.&#x201D; While RAG techniques are intended to reduce hallucinations [<xref ref-type="bibr" rid="ref48">48</xref>], improve content diversity [<xref ref-type="bibr" rid="ref58">58</xref>], user privacy [<xref ref-type="bibr" rid="ref53">53</xref>], and personalization [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref55">55</xref>], our results indicate that the logical reconstruction of fragmented information remains a challenge for RAG-based systems [<xref ref-type="bibr" rid="ref108">108</xref>].</p><p>LLMs can maintain superior structural fluidity and generate well-organized information [<xref ref-type="bibr" rid="ref109">109</xref>]. Our results indicated higher ratings for the general LLM in stylistic criteria, such as structure and communication style. In contrast, we observed that the RAG system&#x2019;s output was inherently tied to multisource retrieval and the informal nature of the retrieved peer-discussion data. However, the practical advantage of our RAG approach lies in its modular architecture, which is specifically designed to facilitate control over response accuracy and prevent the generation of fictitious experiences by relying on verifiable information. Furthermore, our RAG chatbot integrates anonymized real-world peer experiential expertise and Willcox emotional vocabulary to provide a level of cultural authenticity and empathetic alignment that a general-purpose model cannot guarantee. Another advantage of RAG models is their ability to use smaller, localized language models to achieve response quality comparable to significantly larger LLMs [<xref ref-type="bibr" rid="ref110">110</xref>,<xref ref-type="bibr" rid="ref111">111</xref>], thereby providing opportunities to reduce health care barriers related to data privacy and operational costs.</p><p>The participant group expressed a preference for the structured and categorized communication style of the general LLM over the RAG chatbot&#x2019;s response format. This suggests that while RAG can be highly effective in extracting information from diverse sources to provide contextually intelligent responses [<xref ref-type="bibr" rid="ref112">112</xref>], further refinement is needed to synthesize retrieved fragments into a clearer and more structured response format that matches the systematic response delivery of general-purpose LLMs.</p></sec><sec id="s4-4"><title>User Fatigue and Rater Consistency</title><p>Our analysis revealed a residual variance of 0.662 within the linear mixed model, suggesting random noise within the rater data that warrants further investigation. We believe that this inconsistency is primarily attributable to user fatigue stemming from the study&#x2019;s cognitive load. Each participant was required to evaluate 15 complex query sets across 10 distinct criteria, resulting in 450 total data points per rater.</p><p>Furthermore, the high level of domain-specific expertise required to evaluate PrEP support, specifically the distinction between clinical accuracy and nuanced peer experiential support, may have been impacted by the nonexpert and nontarget user status of the participant group. As the raters were neither clinicians nor PrEP candidates, their evaluations of criteria such as &#x201C;accuracy,&#x201D; &#x201C;actionability,&#x201D; &#x201C;clarity,&#x201D; &#x201C;comprehensiveness,&#x201D; &#x201C;detail,&#x201D; &#x201C;empathetic,&#x201D; &#x201C;language,&#x201D; &#x201C;relevance,&#x201D; &#x201C;tailored information,&#x201D; and &#x201C;tone&#x201D; represent perceived credibility rather than objective medical validation. For criteria related to information quality (accuracy, clarity, actionability, and detail), raters lacked the clinical training required to verify the medical correctness of PrEP information and practical guidance. Similarly, evaluating criteria on response relevance and appropriateness (relevance, tailored information, and comprehensiveness) requires an understanding of PrEP challenges such as medical mistrust, PrEP navigation, or side-effect management, commonly experienced by the target population. Furthermore, HIV experts and PrEP candidates are more familiar with PrEP terminology and community lingo, which may impact the perception of tailored information, language, and trust, as these target users can distinguish between generic AI responses and grounded, community-informed experiential support.</p><p>The low incidence of &#x201C;N/A&#x201D; ratings (n=6) suggests that the questionnaire structure may have lacked sufficient clarity in guiding participants to opt out when specific criteria were not relevant to a particular question sample. Improving the questionnaire structure to better highlight the specific intent of each rating and providing more robust instructions will be critical to ensure rater reliability. These exploratory observations will directly inform the next phases of our development, as detailed in the &#x201C;Future Direction&#x201D; section.</p></sec><sec id="s4-5"><title>Implications</title><sec id="s4-5-1"><title>Overview</title><p>RAG chatbots are increasingly used to provide various support for health care use and decision-making [<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref113">113</xref>-<xref ref-type="bibr" rid="ref116">116</xref>]. We report the first published efforts to develop a RAG chatbot prototype for providing personalized information, peer experiential expertise, and emotional support to PrEP candidates. Our chatbot aims to enhance PrEP support by leveraging verified information, real-world experiences, and emotional considerations. It can also interpret complex context-dependent queries, and through personalized informational, peer experiential expertise, and human-like emotional support, we believe this approach has the potential to clarify misconceptions about PrEP and HIV treatment, support user confidence, and facilitate future PrEP uptake.</p></sec><sec id="s4-5-2"><title>Data Privacy and Ethical Implications</title><p>The RAG chatbot currently leverages knowledge from Reddit, with a future plan to expand to include other online platforms. All data usage complies with the respective platforms&#x2019; user agreements and policies, accessed in February 2024, and the facts dataset, collected in January 2022. Recognizing the sensitive nature of health-related discourse in these communities, our data handling process incorporates privacy safeguards; we do not attempt to identify users and protect privacy by filtering personally identifiable information (eg, name, email addresses, IP addresses, and specific unique identifiers).</p><p>The proposed RAG architecture is designed to enhance response precision; however, the deployment of personalized PrEP chatbots carries inherent ethical implications. A primary concern is the probabilistic nature of LLMs. Despite the integration of domain-specific knowledge, these models remain prone to hallucination [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>] and may lack the deep contextual understanding required for nuanced medical inquiries. Consequently, there is a risk of users overrelying on the tool [<xref ref-type="bibr" rid="ref117">117</xref>,<xref ref-type="bibr" rid="ref118">118</xref>] for definitive medical advice, which could delay professional clinical consultation.</p><p>Furthermore, it is critical to distinguish between the cognitive empathy simulated by the LLM and true human-like emotional intelligence. While the system can identify and mirror user sentiments, it lacks the capacity for genuine affective responses [<xref ref-type="bibr" rid="ref43">43</xref>]. This limitation is particularly significant in the context of HIV and PrEP, where users may present with complex psychological distress. The chatbot responses could inadvertently lead to stigma reinforcement if the model misinterprets nuanced emotions and sensitive disclosures or provides inappropriate feedback [<xref ref-type="bibr" rid="ref117">117</xref>,<xref ref-type="bibr" rid="ref118">118</xref>].</p><p>To mitigate these ethical concerns, the RAG architecture implements a modular data governance strategy. The chatbot is designed to use social media data exclusively for addressing experiential queries and mirroring human-like empathetic communication styles while being guided by the Willcox Feeling Wheel to interpret user emotions. Conversely, all factual and clinical queries are strictly routed to the manually curated factual database, ensuring that health-related information is derived solely from verified public health organizations. For transparency, any user content referenced in chatbot responses is quoted anonymously to mitigate the risk of reidentification while maintaining the authenticity of the peer-to-peer perspective.</p></sec></sec><sec id="s4-6"><title>Limitations</title><p>Although recent advances in hallucination removal have improved LLM reliability, these advances do not entirely eliminate the need for validating response accuracy, especially in health care applications [<xref ref-type="bibr" rid="ref92">92</xref>]. Due to linguistic complexity and nuances in generated responses, LLMs struggle to consistently and accurately validate factual information using only RAGs [<xref ref-type="bibr" rid="ref119">119</xref>]. Additional methodological improvements are needed to validate broad facts outside of the database and context-dependent information.</p><p>Our system faces several inherent risks and ethical considerations. First, the complexity of human emotion makes personalized digital support a challenging endeavor, with risks of stigma reinforcement or user overreliance for clinical advice. Second, while using community-sourced data (eg, Reddit) is practical, it may introduce demographic or topical biases that limit the system&#x2019;s generalizability across diverse PrEP user populations. We intend to extend the datasets from MedHelp and POZ communities to mitigate these gaps.</p><p>A limitation of this study is the absence of a comprehensive quantitative analysis comparing the performance of few-shot prompting and fine-tuning for support classification. While our study provides initial qualitative insights into the effectiveness of these 2 approaches for specific tasks, these findings are exploratory. Future work should include extensive empirical comparisons to statistically generalize these results across broader datasets.</p><p>Another limitation of the current evaluation is that the user study was conducted with people who were nontarget PrEP-using demographics and nonclinical experts. While this provides insight into general user perception and readability, it may not capture the technical accuracy or perceived personalization of the responses with precision. Future work will include larger-scale iterative evaluations to identify real-world failure scenarios and ensure the model&#x2019;s safety profile. Furthermore, subsequent iterations will focus on demographic-aware personalization to better reflect the diverse cultural preferences, behaviors, and potential linguistic needs of the PrEP users, addressing the wider ethical implications of using sensitive, public-domain social media data for health support.</p></sec><sec id="s4-7"><title>Future Direction</title><p>The current chatbot model was developed and accessed in a research setting and will be rigorously tested for safety and reliability with experts and the target population prior to real-world deployment. Building on our exploratory findings, we will conduct a larger-scale user study for gathering feedback on chatbot responses from the target population. We will also concurrently initiate a heuristic evaluation with HIV prevention experts to validate clinical safety, user trust, ethical alignment, and factual accuracy of the PrEP chatbot.</p><p>To bridge the gap between initial research product and sustainable real-world health intervention, the chatbot design will incorporate sociotechnical design considerations [<xref ref-type="bibr" rid="ref120">120</xref>] through the Accelerated Creation-to-Sustainment (ACS) framework [<xref ref-type="bibr" rid="ref121">121</xref>]. Through these community-representative user and expert evaluations, we will gather feedback on cultural and behavioral preferences to improve long-term adoption and user trust. This includes incorporating multilingual support to cater to linguistic needs. Furthermore, we will evaluate technological flexibility, including low-cost, lightweight local deployment models and multiplatform accessibility (eg, mobile and web-based interfaces), to lower entry barriers in resource-limited settings. To ensure long-term sustainability, we plan to engage with HIV health stakeholders to establish frameworks for managing scalability and routine content updates, ensuring the intervention remains clinically accurate and relevant as the HIV prevention landscape evolves.</p></sec><sec id="s4-8"><title>Conclusion</title><p>This study details the implementation of a RAG chatbot prototype designed to provide personalized informational, peer experiential expertise, and emotional support to PrEP candidates. By offering personalized information and peer experience, our chatbot can provide essential information for making informed decisions about health. In addition, our chatbot&#x2019;s capability to provide human-like emotional support highlights its potential to address stigma related to PrEP. Although current observations are preliminary, we believe our chatbot has the potential to clarify misconceptions and provide PrEP support through these mechanisms. Future evaluation using expert and real-world feedback will validate and improve the chatbot&#x2019;s capability; after this, a trial to demonstrate the efficacy of the tool to increase PrEP uptake and retention will be proposed.</p></sec></sec></body><back><ack><p>The authors declare the use of Google Gemini (institutional account) strictly for checking grammar and improving sentence structure (STM category 1: editing and proofreading). The tool was not used to generate intellectual content, references, or draft manuscript text. The authors reviewed the final manuscript and are accountable for the integrity and accuracy of the manuscript content.</p></ack><notes><sec><title>Funding</title><p>The authors declare no financial support was received for this work.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ACS</term><def><p>Accelerated Creation-to-Sustainment</p></def></def-item><def-item><term id="abb2">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb3">BLEU</term><def><p>Bilingual Evaluation Understudy</p></def></def-item><def-item><term id="abb4">CN-1</term><def><p>context handling prompt 1</p></def></def-item><def-item><term id="abb5">CN-2</term><def><p>context handling prompt 2</p></def></def-item><def-item><term id="abb6">CR</term><def><p>cohesive response</p></def></def-item><def-item><term id="abb7">EHE</term><def><p>Ending the HIV Epidemic</p></def></def-item><def-item><term id="abb8">IC </term><def><p>information subclassification</p></def></def-item><def-item><term id="abb9">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb10">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb11">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb12">LoRA</term><def><p>Low-Rank Adaptation</p></def></def-item><def-item><term id="abb13">PE</term><def><p>personalized emotion</p></def></def-item><def-item><term id="abb14">PEFT</term><def><p>Parameter-Efficient Fine-Tuning</p></def></def-item><def-item><term id="abb15">PI</term><def><p>personalized information</p></def></def-item><def-item><term id="abb16">PrEP</term><def><p>preexposure prophylaxis</p></def></def-item><def-item><term id="abb17">PX</term><def><p>personalized experience</p></def></def-item><def-item><term id="abb18">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb19">ROUGE-L</term><def><p>Recall-Oriented Understudy for Gisting Evaluation - Longest Common Subsequence</p></def></def-item><def-item><term id="abb20">SAT</term><def><p> Segment Any Text</p></def></def-item><def-item><term id="abb21">SBERT</term><def><p>Sentence-Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb22">SSBC</term><def><p>Social Support Behavior Code</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kroch</surname><given-names>A</given-names> </name><name name-style="western"><surname>O&#x2019;Byrne</surname><given-names>P</given-names> </name><name name-style="western"><surname>Orser</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Increased PrEP uptake and PrEP-RN coincide with decreased HIV diagnoses in men who have sex with men in Ottawa, Canada</article-title><source>Can Commun Dis Rep</source><year>2023</year><month>06</month><day>1</day><volume>49</volume><issue>6</issue><fpage>274</fpage><lpage>281</lpage><pub-id pub-id-type="doi">10.14745/ccdr.v49i06a04</pub-id><pub-id pub-id-type="medline">38440773</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bavinton</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Grulich</surname><given-names>AE</given-names> </name></person-group><article-title>HIV pre-exposure prophylaxis: scaling up for impact now and in the future</article-title><source>Lancet Public Health</source><year>2021</year><month>07</month><volume>6</volume><issue>7</issue><fpage>e528</fpage><lpage>e533</lpage><pub-id pub-id-type="doi">10.1016/S2468-2667(21)00112-2</pub-id><pub-id pub-id-type="medline">34087117</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dimitrov</surname><given-names>DT</given-names> </name><name name-style="western"><surname>M&#x00E2;sse</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Donnell</surname><given-names>D</given-names> </name></person-group><article-title>PrEP adherence patterns strongly affect individual HIV risk and observed efficacy in randomized clinical trials</article-title><source>J Acquir Immune Defic Syndr</source><year>2016</year><month>08</month><day>1</day><volume>72</volume><issue>4</issue><fpage>444</fpage><lpage>451</lpage><pub-id pub-id-type="doi">10.1097/QAI.0000000000000993</pub-id><pub-id pub-id-type="medline">26990823</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="web"><article-title>EHE overview</article-title><source>HIV.gov</source><access-date>2025-04-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.hiv.gov/federal-response/ending-the-hiv-epidemic/overview">https://www.hiv.gov/federal-response/ending-the-hiv-epidemic/overview</ext-link></comment></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sullivan</surname><given-names>PS</given-names> </name><name name-style="western"><surname>DuBose</surname><given-names>SN</given-names> </name><name name-style="western"><surname>Castel</surname><given-names>AD</given-names> </name><etal/></person-group><article-title>Equity of PrEP uptake by race, ethnicity, sex and region in the United States in the first decade of PrEP: a population-based analysis</article-title><source>Lancet Reg Health Am</source><year>2024</year><month>05</month><volume>33</volume><fpage>100738</fpage><pub-id pub-id-type="doi">10.1016/j.lana.2024.100738</pub-id><pub-id pub-id-type="medline">38659491</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mayer</surname><given-names>KH</given-names> </name><name name-style="western"><surname>Agwu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Malebranche</surname><given-names>D</given-names> </name></person-group><article-title>Barriers to the wider use of pre-exposure prophylaxis in the United States: a narrative review</article-title><source>Adv Ther</source><year>2020</year><month>05</month><volume>37</volume><issue>5</issue><fpage>1778</fpage><lpage>1811</lpage><pub-id pub-id-type="doi">10.1007/s12325-020-01295-0</pub-id><pub-id pub-id-type="medline">32232664</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nabunya</surname><given-names>R</given-names> </name><name name-style="western"><surname>Karis</surname><given-names>VMS</given-names> </name><name name-style="western"><surname>Nakanwagi</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Mukisa</surname><given-names>P</given-names> </name><name name-style="western"><surname>Muwanguzi</surname><given-names>PA</given-names> </name></person-group><article-title>Barriers and facilitators to oral PrEP uptake among high-risk men after HIV testing at workplaces in Uganda: a qualitative study</article-title><source>BMC Public Health</source><year>2023</year><month>02</month><day>20</day><volume>23</volume><issue>1</issue><fpage>365</fpage><pub-id pub-id-type="doi">10.1186/s12889-023-15260-3</pub-id><pub-id pub-id-type="medline">36805698</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamoonga</surname><given-names>TE</given-names> </name><name name-style="western"><surname>Mutale</surname><given-names>W</given-names> </name><name name-style="western"><surname>Hill</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Igumbor</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chi</surname><given-names>BH</given-names> </name></person-group><article-title>&#x201C;PrEP protects us&#x201D;: behavioural, normative, and control beliefs influencing pre-exposure prophylaxis uptake among pregnant and breastfeeding women in Zambia</article-title><source>Front Reprod Health</source><year>2023</year><volume>5</volume><fpage>1084657</fpage><pub-id pub-id-type="doi">10.3389/frph.2023.1084657</pub-id><pub-id pub-id-type="medline">37152481</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Crooks</surname><given-names>N</given-names> </name><name name-style="western"><surname>Singer</surname><given-names>RB</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Barriers to PrEP uptake among Black female adolescents and emerging adults</article-title><source>Prev Med Rep</source><year>2023</year><month>02</month><volume>31</volume><fpage>102062</fpage><pub-id pub-id-type="doi">10.1016/j.pmedr.2022.102062</pub-id><pub-id pub-id-type="medline">36467542</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Conner</surname><given-names>KO</given-names> </name><name name-style="western"><surname>McKinnon</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Reynolds</surname><given-names>CF</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>C</given-names> </name></person-group><article-title>Peer education as a strategy for reducing internalized stigma among depressed older adults</article-title><source>Psychiatr Rehabil J</source><year>2015</year><month>06</month><volume>38</volume><issue>2</issue><fpage>186</fpage><lpage>193</lpage><pub-id pub-id-type="doi">10.1037/prj0000109</pub-id><pub-id pub-id-type="medline">25915057</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>X</given-names> </name></person-group><article-title>How could peers in online health community help improve health behavior</article-title><source>Int J Environ Res Public Health</source><year>2020</year><month>01</month><volume>17</volume><issue>9</issue><fpage>2995</fpage><pub-id pub-id-type="doi">10.3390/ijerph17092995</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deci</surname><given-names>EL</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>RM</given-names> </name></person-group><article-title>The &#x201C;What&#x201D; and &#x201C;Why&#x201D; of goal pursuits: human needs and the self-determination of behavior</article-title><source>Psychol Inq</source><year>2000</year><access-date>2026-08-12</access-date><volume>11</volume><issue>4</issue><fpage>227</fpage><lpage>268</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.tandfonline.com/doi/abs/10.1207/s15327965pli1104_01">https://www.tandfonline.com/doi/abs/10.1207/s15327965pli1104_01</ext-link></comment><pub-id pub-id-type="doi">10.1207/S15327965PLI1104_01</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burke</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pyle</surname><given-names>M</given-names> </name><name name-style="western"><surname>Machin</surname><given-names>K</given-names> </name><name name-style="western"><surname>Varese</surname><given-names>F</given-names> </name><name name-style="western"><surname>Morrison</surname><given-names>AP</given-names> </name></person-group><article-title>The effects of peer support on empowerment, self-efficacy, and internalized stigma: a narrative synthesis and meta-analysis</article-title><source>Stigma Health</source><year>2019</year><volume>4</volume><issue>3</issue><fpage>337</fpage><lpage>356</lpage><pub-id pub-id-type="doi">10.1037/sah0000148</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>J</given-names> </name><name name-style="western"><surname>Killelea</surname><given-names>A</given-names> </name><name name-style="western"><surname>Farrow</surname><given-names>K</given-names> </name></person-group><article-title>Investing in national HIV PrEP preparedness</article-title><source>N Engl J Med</source><year>2023</year><month>03</month><day>2</day><volume>388</volume><issue>9</issue><fpage>769</fpage><lpage>771</lpage><pub-id pub-id-type="doi">10.1056/NEJMp2216100</pub-id><pub-id pub-id-type="medline">36847476</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hill</surname><given-names>M</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>J</given-names> </name><name name-style="western"><surname>Elimam</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Ending the HIV epidemic PrEP equity recommendations from a rapid ethnographic assessment of multilevel PrEP use determinants among young Black gay and bisexual men in Atlanta, GA</article-title><source>PLoS One</source><year>2023</year><volume>18</volume><issue>3</issue><fpage>e0283764</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0283764</pub-id><pub-id pub-id-type="medline">36996143</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sullivan</surname><given-names>PS</given-names> </name><name name-style="western"><surname>Siegler</surname><given-names>AJ</given-names> </name></person-group><article-title>Getting pre-exposure prophylaxis (PrEP) to the people: opportunities, challenges and emerging models of PrEP implementation</article-title><source>Sex Health</source><year>2018</year><month>11</month><volume>15</volume><issue>6</issue><fpage>522</fpage><lpage>527</lpage><pub-id pub-id-type="doi">10.1071/SH18103</pub-id><pub-id pub-id-type="medline">30476461</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maita</surname><given-names>KC</given-names> </name><name name-style="western"><surname>Maniaci</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Haider</surname><given-names>CR</given-names> </name><etal/></person-group><article-title>The impact of digital health solutions on bridging the health care gap in rural areas: a scoping review</article-title><source>Perm J</source><year>2024</year><month>09</month><day>16</day><volume>28</volume><issue>3</issue><fpage>130</fpage><lpage>143</lpage><pub-id pub-id-type="doi">10.7812/TPP/23.134</pub-id><pub-id pub-id-type="medline">39135461</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Coleman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bojan</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Developing a mobile app (LYNX) to support linkage to HIV/sexually transmitted infection testing and pre-exposure prophylaxis for young men who have sex with men: protocol for a randomized controlled trial</article-title><source>JMIR Res Protoc</source><year>2019</year><month>01</month><day>25</day><volume>8</volume><issue>1</issue><fpage>e10659</fpage><pub-id pub-id-type="doi">10.2196/10659</pub-id><pub-id pub-id-type="medline">30681964</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Braddock</surname><given-names>WRT</given-names> </name><name name-style="western"><surname>Ocasio</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Comulada</surname><given-names>WS</given-names> </name><name name-style="western"><surname>Mandani</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fernandez</surname><given-names>MI</given-names> </name></person-group><article-title>Increasing participation in a telePrEP program for sexual and gender minority adolescents and young adults in louisiana: protocol for an SMS text messaging-based chatbot</article-title><source>JMIR Res Protoc</source><year>2023</year><month>05</month><day>31</day><volume>12</volume><issue>1</issue><fpage>e42983</fpage><pub-id pub-id-type="doi">10.2196/42983</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ntinga</surname><given-names>X</given-names> </name><name name-style="western"><surname>Musiello</surname><given-names>F</given-names> </name><name name-style="western"><surname>Keter</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Barnabas</surname><given-names>R</given-names> </name><name name-style="western"><surname>van Heerden</surname><given-names>A</given-names> </name></person-group><article-title>The feasibility and acceptability of an mHealth conversational agent designed to support HIV self-testing in South Africa: cross-sectional study</article-title><source>J Med Internet Res</source><year>2022</year><month>12</month><day>12</day><volume>24</volume><issue>12</issue><fpage>e39816</fpage><pub-id pub-id-type="doi">10.2196/39816</pub-id><pub-id pub-id-type="medline">36508248</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>CK</given-names> </name><etal/></person-group><article-title>Evaluating an innovative HIV self-testing service with web-based, real-time counseling provided by an artificial intelligence chatbot (HIVST-chatbot) in increasing HIV self-testing use among Chinese men who have sex with men: protocol for a noninferiority randomized controlled trial</article-title><source>JMIR Res Protoc</source><year>2023</year><month>06</month><day>30</day><volume>12</volume><fpage>e48447</fpage><pub-id pub-id-type="doi">10.2196/48447</pub-id><pub-id pub-id-type="medline">37389935</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>AY</given-names> </name><name name-style="western"><surname>Alleyne</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Doblecki-Lewis</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Adapting mHealth interventions (PrEPmate and DOT diary) to support PrEP retention in care and adherence among English and Spanish-speaking men who have sex with men and transgender women in the United States: formative work and pilot randomized trial</article-title><source>JMIR Form Res</source><year>2024</year><month>03</month><day>27</day><volume>8</volume><issue>1</issue><fpage>e54073</fpage><pub-id pub-id-type="doi">10.2196/54073</pub-id><pub-id pub-id-type="medline">38536232</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Massa</surname><given-names>P</given-names> </name><name name-style="western"><surname>de Souza Ferraz</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Magno</surname><given-names>L</given-names> </name><etal/></person-group><article-title>A transgender chatbot (Amanda Selfie) to create pre-exposure prophylaxis demand among adolescents in Brazil: assessment of acceptability, functionality, usability, and results</article-title><source>J Med Internet Res</source><year>2023</year><month>06</month><day>23</day><volume>25</volume><fpage>e41881</fpage><pub-id pub-id-type="doi">10.2196/41881</pub-id><pub-id pub-id-type="medline">37351920</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wharton</surname><given-names>M</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name></person-group><article-title>Ameliorating racial disparities in HIV prevention via a nurse-led, AI-enhanced program for pre-exposure prophylaxis utilization among black cisgender women: protocol for a mixed methods study</article-title><source>JMIR Res Protoc</source><year>2024</year><month>08</month><day>13</day><volume>13</volume><issue>1</issue><fpage>e59975</fpage><pub-id pub-id-type="doi">10.2196/59975</pub-id><pub-id pub-id-type="medline">39137028</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Muessig</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Knudtson</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Soni</surname><given-names>K</given-names> </name><etal/></person-group><article-title>&#x201C;I didn&#x2019;t tell you sooner because i didn&#x2019;t know how to handle it myself.&#x201D; Developing a virtual reality program to support HIV-status disclosure decisions</article-title><source>Digit Cult Educ</source><year>2018</year><volume>10</volume><fpage>22</fpage><lpage>48</lpage><pub-id pub-id-type="medline">30123342</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hightow-Weidman</surname><given-names>LB</given-names> </name><name name-style="western"><surname>Muessig</surname><given-names>K</given-names> </name><name name-style="western"><surname>Soberano</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Tough talks virtual simulation HIV disclosure intervention for young men who have sex with men: development and usability testing</article-title><source>JMIR Form Res</source><year>2022</year><month>09</month><day>8</day><volume>6</volume><issue>9</issue><fpage>e38354</fpage><pub-id pub-id-type="doi">10.2196/38354</pub-id><pub-id pub-id-type="medline">36074551</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Galea</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Vasquez</surname><given-names>DH</given-names> </name><name name-style="western"><surname>Rupani</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Development and pilot-testing of an optimized conversational agent or &#x201C;Chatbot&#x201D; for peruvian adolescents living with HIV to facilitate mental health screening, education, self-help, and linkage to care: protocol for a mixed methods, community-engaged study</article-title><source>JMIR Res Protoc</source><year>2024</year><month>05</month><day>7</day><volume>13</volume><fpage>e55559</fpage><pub-id pub-id-type="doi">10.2196/55559</pub-id><pub-id pub-id-type="medline">38713501</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sanders</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name><name name-style="western"><surname>Chow</surname><given-names>JCL</given-names> </name></person-group><article-title>Chatbot for health care and oncology applications using artificial intelligence and machine learning: systematic review</article-title><source>JMIR Cancer</source><year>2021</year><month>11</month><day>29</day><volume>7</volume><issue>4</issue><fpage>e27850</fpage><pub-id pub-id-type="doi">10.2196/27850</pub-id><pub-id pub-id-type="medline">34847056</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ardiana</surname><given-names>DPY</given-names> </name><name name-style="western"><surname>Joni</surname><given-names>IDMAB</given-names> </name><name name-style="western"><surname>Udayana</surname><given-names>IPAED</given-names> </name></person-group><article-title>Mobile based chatbot application for HIV/AIDS counseling using artificial intelligence markup language approach</article-title><source>J Phys: Conf Ser</source><year>2020</year><month>02</month><day>1</day><volume>1469</volume><issue>1</issue><fpage>012041</fpage><pub-id pub-id-type="doi">10.1088/1742-6596/1469/1/012041</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moreno</surname><given-names>JC</given-names> </name><name name-style="western"><surname>S&#x00E1;nchez-Anguix</surname><given-names>V</given-names> </name><name name-style="western"><surname>Alberola</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Juli&#x00E1;n</surname><given-names>V</given-names> </name><name name-style="western"><surname>Botti</surname><given-names>V</given-names> </name></person-group><article-title>An intelligent conversational agent for educating the general public about HIV</article-title><source>Neurocomputing</source><year>2024</year><month>01</month><volume>563</volume><fpage>126902</fpage><pub-id pub-id-type="doi">10.1016/j.neucom.2023.126902</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hauser-Ulrich</surname><given-names>S</given-names> </name><name name-style="western"><surname>K&#x00FC;nzli</surname><given-names>H</given-names> </name><name name-style="western"><surname>Meier-Peterhans</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kowatsch</surname><given-names>T</given-names> </name></person-group><article-title>A smartphone-based health care chatbot to promote self-management of chronic pain (SELMA): pilot randomized controlled trial</article-title><source>JMIR Mhealth Uhealth</source><year>2020</year><month>04</month><day>3</day><volume>8</volume><issue>4</issue><fpage>e15806</fpage><pub-id pub-id-type="doi">10.2196/15806</pub-id><pub-id pub-id-type="medline">32242820</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Oliver</surname><given-names>D</given-names> </name><name name-style="western"><surname>Msosa</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Implementation of a real-time psychosis risk detection and alerting system based on electronic health records using CogStack</article-title><source>J Vis Exp</source><year>2020</year><month>05</month><day>15</day><volume>PMID</volume><issue>159</issue><pub-id pub-id-type="doi">10.3791/60794</pub-id><pub-id pub-id-type="medline">32478737</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fulmer</surname><given-names>R</given-names> </name><name name-style="western"><surname>Joerin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gentile</surname><given-names>B</given-names> </name><name name-style="western"><surname>Lakerink</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rauws</surname><given-names>M</given-names> </name></person-group><article-title>Using psychological artificial intelligence (Tess) to relieve symptoms of depression and anxiety: randomized controlled trial</article-title><source>JMIR Ment Health</source><year>2018</year><month>12</month><day>13</day><volume>5</volume><issue>4</issue><fpage>e64</fpage><pub-id pub-id-type="doi">10.2196/mental.9782</pub-id><pub-id pub-id-type="medline">30545815</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>H</given-names> </name><name name-style="western"><surname>Song</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>M</given-names> </name></person-group><article-title>Using AI chatbots to provide self-help depression interventions for university students: a randomized trial of effectiveness</article-title><source>Internet Interv</source><year>2022</year><month>03</month><volume>27</volume><fpage>100495</fpage><pub-id pub-id-type="doi">10.1016/j.invent.2022.100495</pub-id><pub-id pub-id-type="medline">35059305</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>E</given-names> </name></person-group><article-title>Mobile app-based chatbot to deliver cognitive behavioral therapy and psychoeducation for adults with attention deficit: a development and feasibility/usability study</article-title><source>Int J Med Inform</source><year>2021</year><month>06</month><volume>150</volume><fpage>104440</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2021.104440</pub-id><pub-id pub-id-type="medline">33799055</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamid</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Valicevic</surname><given-names>A</given-names> </name><name name-style="western"><surname>Brenneman</surname><given-names>B</given-names> </name><name name-style="western"><surname>Niziol</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Stein</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Newman-Casey</surname><given-names>PA</given-names> </name></person-group><article-title>Text parsing-based identification of patients with poor glaucoma medication adherence in the electronic health record</article-title><source>Am J Ophthalmol</source><year>2021</year><month>02</month><volume>222</volume><fpage>54</fpage><lpage>59</lpage><pub-id pub-id-type="doi">10.1016/j.ajo.2020.09.008</pub-id><pub-id pub-id-type="medline">32926847</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>De Nieva</surname><given-names>JO</given-names> </name><name name-style="western"><surname>Joaquin</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>CB</given-names> </name><name name-style="western"><surname>Marc Te</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Ong</surname><given-names>E</given-names> </name></person-group><article-title>Investigating students&#x2019; use of a mental health chatbot to alleviate academic stress</article-title><access-date>2026-08-12</access-date><conf-name>CHIuXiD &#x2019;20: 6th International ACM In-Cooperation HCI and UX Conference</conf-name><conf-date>Oct 21-23, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3431656">https://dl.acm.org/doi/proceedings/10.1145/3431656</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Savova</surname><given-names>GK</given-names> </name><name name-style="western"><surname>Masanz</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Ogren</surname><given-names>PV</given-names> </name><etal/></person-group><article-title>Mayo clinical Text Analysis and Knowledge Extraction System (cTAKES): architecture, component evaluation and applications</article-title><source>J Am Med Inform Assoc</source><year>2010</year><volume>17</volume><issue>5</issue><fpage>507</fpage><lpage>513</lpage><pub-id pub-id-type="doi">10.1136/jamia.2009.001560</pub-id><pub-id pub-id-type="medline">20819853</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ali</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Razavi</surname><given-names>SZ</given-names> </name><name name-style="western"><surname>Langevin</surname><given-names>R</given-names> </name><etal/></person-group><article-title>A virtual conversational agent for teens with autism spectrum disorder: experimental results and design lessons</article-title><access-date>2026-09-19</access-date><conf-name>IVA &#x2019;20: 20th ACM International Conference on Intelligent Virtual Agents</conf-name><conf-date>Oct 19-23, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.1145/3383652.3423900">https://dl.acm.org/doi/10.1145/3383652.3423900</ext-link></comment><pub-id pub-id-type="doi">10.1145/3383652.3423900</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peng</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Wickersham</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Altice</surname><given-names>FL</given-names> </name><etal/></person-group><article-title>Formative evaluation of the acceptance of HIV prevention artificial intelligence chatbots by men who have sex with men in Malaysia: focus group study</article-title><source>JMIR Form Res</source><year>2022</year><month>10</month><day>6</day><volume>6</volume><issue>10</issue><fpage>e42055</fpage><pub-id pub-id-type="doi">10.2196/42055</pub-id><pub-id pub-id-type="medline">36201390</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nadarzynski</surname><given-names>T</given-names> </name><name name-style="western"><surname>Puentes</surname><given-names>V</given-names> </name><name name-style="western"><surname>Pawlak</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Barriers and facilitators to engagement with artificial intelligence (AI)-based chatbots for sexual and reproductive health advice: a qualitative analysis</article-title><source>Sex Health</source><year>2021</year><month>11</month><volume>18</volume><issue>5</issue><fpage>385</fpage><lpage>393</lpage><pub-id pub-id-type="doi">10.1071/SH21123</pub-id><pub-id pub-id-type="medline">34782055</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>M</given-names> </name><name name-style="western"><surname>Warren</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Abdul-Muhsin</surname><given-names>HM</given-names> </name><name name-style="western"><surname>Banerjee</surname><given-names>I</given-names> </name></person-group><article-title>Assessing empathy in large language models with real-world physician-patient interactions</article-title><year>2024</year><access-date>2026-09-19</access-date><conf-name>2024 IEEE International Conference on Big Data (BigData)</conf-name><conf-date>Dec 15-18, 2024</conf-date><conf-loc>Washington, DC, USA</conf-loc><fpage>6510</fpage><lpage>6519</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/document/10825307">https://ieeexplore.ieee.org/document/10825307</ext-link></comment><pub-id pub-id-type="doi">10.1109/BigData62323.2024.10825307</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vzorin</surname><given-names>GD</given-names> </name><name name-style="western"><surname>Bukinich</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Sedykh</surname><given-names>AV</given-names> </name><name name-style="western"><surname>Vetrova</surname><given-names>II</given-names> </name><name name-style="western"><surname>Sergienko</surname><given-names>EA</given-names> </name></person-group><article-title>The emotional intelligence of the GPT-4 large language model</article-title><source>Psychol Russ</source><year>2024</year><volume>17</volume><issue>2</issue><fpage>85</fpage><lpage>99</lpage><pub-id pub-id-type="doi">10.11621/pir.2024.0206</pub-id><pub-id pub-id-type="medline">39552777</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Brin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Barash</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Large language models and empath: systematic review</article-title><source>J Med Internet Res</source><year>2024</year><month>12</month><volume>26</volume><issue>1</issue><fpage>e52597</fpage><pub-id pub-id-type="doi">10.2196/52597</pub-id><pub-id pub-id-type="medline">39661968</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Montemayor</surname><given-names>C</given-names> </name><name name-style="western"><surname>Halpern</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fairweather</surname><given-names>A</given-names> </name></person-group><article-title>In principle obstacles for empathic AI: why we can&#x2019;t replace human empathy in healthcare</article-title><source>AI Soc</source><year>2022</year><volume>37</volume><issue>4</issue><fpage>1353</fpage><lpage>1359</lpage><pub-id pub-id-type="doi">10.1007/s00146-021-01230-z</pub-id><pub-id pub-id-type="medline">34054228</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>HQ</given-names> </name><name name-style="western"><surname>McGuinness</surname><given-names>S</given-names> </name></person-group><article-title>An experimental study of integrating fine-tuned large language models and prompts for enhancing mental health support chatbot system</article-title><source>J Med Artif Intell</source><year>2024</year><month>06</month><day>4</day><volume>7</volume><fpage>16</fpage><lpage>16</lpage><pub-id pub-id-type="doi">10.21037/jmai-23-136</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ye</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hua</surname><given-names>W</given-names> </name><name name-style="western"><surname>Jia</surname><given-names>W</given-names> </name></person-group><article-title>Cognitive mirage: a review of hallucinations in large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 13, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2309.06794</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gilson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ai</surname><given-names>X</given-names> </name><name name-style="western"><surname>Arunachalam</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Enhancing large language models with domain-specific retrieval augment generation: a case study on long-form consumer health question answering in ophthalmology</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 20, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2409.13902</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Prompt engineering for healthcare: methodologies and applications</article-title><source>Meta-Radiology</source><year>2026</year><month>03</month><volume>4</volume><issue>1</issue><fpage>100190</fpage><pub-id pub-id-type="doi">10.1016/j.metrad.2025.100190</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burford</surname><given-names>KG</given-names> </name><name name-style="western"><surname>Itzkowitz</surname><given-names>NG</given-names> </name><name name-style="western"><surname>Ortega</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Teitler</surname><given-names>JO</given-names> </name><name name-style="western"><surname>Rundle</surname><given-names>AG</given-names> </name></person-group><article-title>Use of generative AI to identify helmet status among patients with micromobility-related injuries from unstructured clinical notes</article-title><source>JAMA Netw Open</source><year>2024</year><month>08</month><day>1</day><volume>7</volume><issue>8</issue><fpage>e2425981</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.25981</pub-id><pub-id pub-id-type="medline">39136946</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guevara</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Large language models to identify social determinants of health in electronic health records</article-title><source>NPJ Digit Med</source><year>2024</year><month>01</month><day>11</day><volume>7</volume><issue>1</issue><fpage>6</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00970-0</pub-id><pub-id pub-id-type="medline">38200151</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>P</given-names> </name><name name-style="western"><surname>Sano</surname><given-names>A</given-names> </name></person-group><article-title>Zero-shot ECG diagnosis with large language models and retrieval-augmented generation</article-title><year>2023</year><conf-name>Proceedings of the 3rd Machine Learning for Health Symposium, PMLR</conf-name><fpage>650</fpage><lpage>663</lpage></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zeng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>He</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Mitigating the privacy issues in retrieval-augmented generation (RAG) via pure synthetic data</article-title><access-date>2026-08-12</access-date><conf-name>2025 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 4-7, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.emnlp-main">https://aclanthology.org/2025.emnlp-main</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2025.emnlp-main.1247</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Salemi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mysore</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bendersky</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zamani</surname><given-names>H</given-names> </name></person-group><article-title>LaMP: when large language models meet personalization</article-title><access-date>2026-09-19</access-date><conf-name>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</conf-name><conf-date>Aug 11-16, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.acl-long.399/">https://aclanthology.org/2024.acl-long.399/</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2024.acl-long.399</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>When large language models meet personalization: perspectives of challenges and opportunities</article-title><source>World Wide Web</source><year>2024</year><month>07</month><volume>27</volume><issue>4</issue><fpage>42</fpage><pub-id pub-id-type="doi">10.1007/s11280-024-01276-1</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Shi</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhuang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Iwinski</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wattenbarger</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>MD</given-names> </name></person-group><article-title>Retrieval-augmented large language models for adolescent idiopathic scoliosis patients in shared decision-making</article-title><access-date>2026-09-19</access-date><conf-name>BCB&#x2019;23: 14th ACM International Conference on Bioinformatics, Computational Biology, and Health Informatics</conf-name><conf-date>Sep 3-6, 2023</conf-date><conf-loc>Houston, TX</conf-loc><fpage>1</fpage><lpage>10</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3584371">https://dl.acm.org/doi/proceedings/10.1145/3584371</ext-link></comment><pub-id pub-id-type="doi">10.1145/3584371.3612956</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ramjee</surname><given-names>P</given-names> </name><name name-style="western"><surname>Sachdeva</surname><given-names>B</given-names> </name><name name-style="western"><surname>Golechha</surname><given-names>S</given-names> </name><etal/></person-group><article-title>CataractBot: an LLM-powered expert-in-the-loop chatbot for cataract patients</article-title><source>Proc ACM Interact Mob Wearable Ubiquitous Technol</source><year>2025</year><month>06</month><day>9</day><volume>9</volume><issue>2</issue><fpage>1</fpage><lpage>31</lpage><pub-id pub-id-type="doi">10.1145/3729479</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ng</surname><given-names>KKY</given-names> </name><name name-style="western"><surname>Matsuba</surname><given-names>I</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>PC</given-names> </name></person-group><article-title>RAG in health care: a novel framework for improving communication and decision-making by addressing LLM limitations</article-title><source>NEJM AI</source><year>2025</year><month>01</month><volume>2</volume><issue>1</issue><fpage>AIra2400380</fpage><pub-id pub-id-type="doi">10.1056/AIra2400380</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Duan</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>GastroBot: a Chinese gastrointestinal disease chatbot based on the retrieval-augmented generation</article-title><source>Front Med (Lausanne)</source><year>2024</year><month>05</month><day>22</day><volume>11</volume><fpage>1392555</fpage><pub-id pub-id-type="doi">10.3389/fmed.2024.1392555</pub-id><pub-id pub-id-type="medline">38841582</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gordillo</surname><given-names>V</given-names> </name><name name-style="western"><surname>Fekete</surname><given-names>E</given-names> </name><name name-style="western"><surname>Platteau</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Emotional support and gender in people living with HIV: effects on psychological well-being</article-title><source>J Behav Med</source><year>2009</year><month>12</month><volume>32</volume><issue>6</issue><fpage>523</fpage><lpage>531</lpage><pub-id pub-id-type="doi">10.1007/s10865-009-9222-7</pub-id><pub-id pub-id-type="medline">19543823</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smit</surname><given-names>F</given-names> </name><name name-style="western"><surname>Masvawure</surname><given-names>TB</given-names> </name></person-group><article-title>Barriers and facilitators to acceptability and uptake of pre-exposure prophylaxis (PrEP) among Black women in the United States: a systematic review</article-title><source>J Racial and Ethnic Health Disparities</source><year>2024</year><month>10</month><volume>11</volume><issue>5</issue><fpage>2649</fpage><lpage>2662</lpage><pub-id pub-id-type="doi">10.1007/s40615-023-01729-9</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Duthely</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Sanchez-Covarrubias</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>MR</given-names> </name><etal/></person-group><article-title>Pills, PrEP, and Pals: adherence, stigma, resilience, faith and the need to connect among minority women with HIV/AIDS in a US HIV epicenter</article-title><source>Front Public Health</source><year>2021</year><volume>9</volume><fpage>667331</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2021.667331</pub-id><pub-id pub-id-type="medline">34235129</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hartzler</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pratt</surname><given-names>W</given-names> </name></person-group><article-title>Managing the personal side of health: how patient expertise differs from the expertise of clinicians</article-title><source>J Med Internet Res</source><year>2011</year><month>08</month><day>16</day><volume>13</volume><issue>3</issue><fpage>e62</fpage><pub-id pub-id-type="doi">10.2196/jmir.1728</pub-id><pub-id pub-id-type="medline">21846635</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dennis</surname><given-names>CL</given-names> </name></person-group><article-title>Peer support within a health care context: a concept analysis</article-title><source>Int J Nurs Stud</source><year>2003</year><month>03</month><volume>40</volume><issue>3</issue><fpage>321</fpage><lpage>332</lpage><pub-id pub-id-type="doi">10.1016/s0020-7489(02)00092-5</pub-id><pub-id pub-id-type="medline">12605954</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>YC</given-names> </name><name name-style="western"><surname>Kraut</surname><given-names>R</given-names> </name><name name-style="western"><surname>Levine</surname><given-names>JM</given-names> </name></person-group><article-title>To stay or leave? the relationship of emotional and informational support to commitment in online health support groups</article-title><access-date>2026-09-19</access-date><conf-name>CSCW&#x2019;12: Proceedings of the ACM Conference on Computer Supported Cooperative Work</conf-name><conf-date>Feb 11-15, 2012</conf-date><conf-loc>Seattle, WA</conf-loc><fpage>833</fpage><lpage>842</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.1145/2145204.2145329">https://dl.acm.org/doi/10.1145/2145204.2145329</ext-link></comment><pub-id pub-id-type="doi">10.1145/2145204.2145329</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Castro</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Van Regenmortel</surname><given-names>T</given-names> </name><name name-style="western"><surname>Sermeus</surname><given-names>W</given-names> </name><name name-style="western"><surname>Vanhaecht</surname><given-names>K</given-names> </name></person-group><article-title>Patients&#x2019; experiential knowledge and expertise in health care: a hybrid concept analysis</article-title><source>Soc Theory Health</source><year>2019</year><month>09</month><volume>17</volume><issue>3</issue><fpage>307</fpage><lpage>330</lpage><pub-id pub-id-type="doi">10.1057/s41285-018-0081-6</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Burleson</surname><given-names>BR</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Greene</surname><given-names>JO</given-names> </name><name name-style="western"><surname>Burleson</surname><given-names>BR</given-names> </name></person-group><article-title>Emotional support skills</article-title><source>Handbook of Communication and Social Interaction Skills Routledge</source><year>2003</year><publisher-name>Lawrence Erlbaum Associates, Inc</publisher-name><pub-id pub-id-type="doi">10.4324/9781410607133-22</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Plutchik</surname><given-names>R</given-names> </name></person-group><article-title>The nature of emotions: human emotions have deep evolutionary roots, a fact that may explain their complexity and provide tools for clinical practice</article-title><source>Am Sci</source><year>2001</year><volume>89</volume><issue>4</issue><fpage>344</fpage><lpage>350</lpage><pub-id pub-id-type="doi">10.1511/2001.28.344</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Willcox</surname><given-names>G</given-names> </name></person-group><article-title>The feeling wheel: a tool for expanding awareness of emotions and increasing spontaneity and intimacy</article-title><source>Trans Anal J</source><year>1982</year><month>10</month><day>1</day><volume>12</volume><issue>4</issue><fpage>274</fpage><lpage>276</lpage><pub-id pub-id-type="doi">10.1177/036215378201200411</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chu</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Viswanath</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lam</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>SSC</given-names> </name></person-group><article-title>How, when and why people seek health information online: qualitative study in Hong Kong</article-title><source>Interact J Med Res</source><year>2017</year><month>12</month><day>12</day><volume>6</volume><issue>2</issue><fpage>e24</fpage><pub-id pub-id-type="doi">10.2196/ijmr.7000</pub-id><pub-id pub-id-type="medline">29233802</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Aliannejadi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chakraborty</surname><given-names>M</given-names> </name><name name-style="western"><surname>R&#x00ED;ssola</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Crestani</surname><given-names>F</given-names> </name></person-group><article-title>Harnessing evolution of multi-turn conversations for effective answer retrieval</article-title><access-date>2026-08-12</access-date><conf-name>CHIR &#x2019;20: Conference on Human Information Interaction and Retrieval</conf-name><conf-date>Aug 14-18, 2020</conf-date><conf-loc>Vancouver, BC, Canada</conf-loc><fpage>33</fpage><lpage>42</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3343413">https://dl.acm.org/doi/proceedings/10.1145/3343413</ext-link></comment><pub-id pub-id-type="doi">10.1145/3343413.3377968</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Strine</surname><given-names>TW</given-names> </name><name name-style="western"><surname>Chapman</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Balluz</surname><given-names>L</given-names> </name><name name-style="western"><surname>Mokdad</surname><given-names>AH</given-names> </name></person-group><article-title>Health-related quality of life and health behaviors by social and emotional support</article-title><source>Soc Psychiat Epidemiol</source><year>2008</year><month>02</month><volume>43</volume><issue>2</issue><fpage>151</fpage><lpage>159</lpage><pub-id pub-id-type="doi">10.1007/s00127-007-0277-x</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ye</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Enhancement of the performance of large language models in diabetes education through retrieval-augmented generation: comparative study</article-title><source>J Med Internet Res</source><year>2024</year><month>11</month><day>8</day><volume>26</volume><fpage>e58041</fpage><pub-id pub-id-type="doi">10.2196/58041</pub-id><pub-id pub-id-type="medline">39046096</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cutrona</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Suhr</surname><given-names>JA</given-names> </name></person-group><article-title>Controllability of stressful events and satisfaction with spouse support behaviors</article-title><source>Communic Res</source><year>1992</year><month>04</month><volume>19</volume><issue>2</issue><fpage>154</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.1177/009365092019002002</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Suhr</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Cutrona</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Krebs</surname><given-names>KK</given-names> </name><name name-style="western"><surname>Jensen</surname><given-names>SL</given-names> </name></person-group><article-title>The social support behavior code (SSBC)</article-title><source>Couple Observational Coding Systems</source><year>2004</year><access-date>2026-08-10</access-date><publisher-name>Routedge</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.taylorfrancis.com/chapters/edit/10.4324/9781410610843-24/social-support-behavior-code-ssbc-julie-suhr-carolyn-cutrona-krista-krebs-sandra-jensen">https://www.taylorfrancis.com/chapters/edit/10.4324/9781410610843-24/social-support-behavior-code-ssbc-julie-suhr-carolyn-cutrona-krista-krebs-sandra-jensen</ext-link></comment></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rousseau</surname><given-names>E</given-names> </name><name name-style="western"><surname>Katz</surname><given-names>AWK</given-names> </name><name name-style="western"><surname>O&#x2019;Rourke</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Adolescent girls and young women&#x2019;s PrEP-user journey during an implementation science study in South Africa and Kenya</article-title><source>PLoS One</source><year>2021</year><volume>16</volume><issue>10</issue><fpage>e0258542</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0258542</pub-id><pub-id pub-id-type="medline">34648589</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Frohmann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sterner</surname><given-names>I</given-names> </name><name name-style="western"><surname>Vuli&#x0107;</surname><given-names>I</given-names> </name><name name-style="western"><surname>Minixhofer</surname><given-names>B</given-names> </name><name name-style="western"><surname>Schedl</surname><given-names>M</given-names> </name></person-group><article-title>Segment any text: a universal approach for robust, efficient and adaptable sentence segmentation</article-title><access-date>2026-06-19</access-date><conf-name>Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.emnlp-main.665/">https://aclanthology.org/2024.emnlp-main.665/</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.665</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Team</surname><given-names>G</given-names> </name><name name-style="western"><surname>Riviere</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pathak</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Gemma 2: improving open language models at a practical size</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 2, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.00118</pub-id></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lialin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Deshpande</surname><given-names>V</given-names> </name><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name></person-group><article-title>Scaling down to scale up: a guide to parameter-efficient fine-tuning</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 28, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.15647</pub-id></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Malladi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>T</given-names> </name><name name-style="western"><surname>Nichani</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Fine-tuning language models with just forward passes</article-title><access-date>2026-09-19</access-date><conf-name>The 37th Annual Conference on Neural Information Processing Systems (NeurIPS 2023)</conf-name><conf-date>Dec 10-16, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2023/file/a627810151be4d13f907ac898ff7e948-Paper-Conference.pdf?">https://proceedings.neurips.cc/paper_files/paper/2023/file/a627810151be4d13f907ac898ff7e948-Paper-Conference.pdf?</ext-link></comment><pub-id pub-id-type="doi">10.52202/075280-2308</pub-id></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Maatouk</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ampudia</surname><given-names>KC</given-names> </name><name name-style="western"><surname>Ying</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tassiulas</surname><given-names>L</given-names> </name></person-group><article-title>Tele-LLMs: a series of specialized large language models for telecommunications</article-title><source>arXiv</source><access-date>2026-09-21</access-date><comment>Preprint posted online on  Sep 9, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2409.05314</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name></person-group><article-title>Sentence-BERT: sentence embeddings using siamese BERT-networks</article-title><access-date>2026-08-12</access-date><conf-name>2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-date>Nov 3-7, 2019</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.aclweb.org/anthology/D19-1">https://www.aclweb.org/anthology/D19-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D19-1410</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mahajan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bansal</surname><given-names>N</given-names> </name><name name-style="western"><surname>Blanco</surname><given-names>E</given-names> </name><name name-style="western"><surname>Karmaker</surname><given-names>S</given-names> </name></person-group><article-title>ALIGN-SIM: a task-free test bed for evaluating and interpreting sentence embeddings through semantic similarity alignment</article-title><access-date>2026-09-19</access-date><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Nov 12-16, 2024</conf-date><conf-loc>Miami, Florida, USA</conf-loc><fpage>7393</fpage><lpage>7428</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.findings-emnlp.436/">https://aclanthology.org/2024.findings-emnlp.436/</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2024.findings-emnlp.436</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tamine</surname><given-names>L</given-names> </name><name name-style="western"><surname>Goeuriot</surname><given-names>L</given-names> </name></person-group><article-title>Semantic information retrieval on medical texts</article-title><source>ACM Comput Surv</source><year>2022</year><month>09</month><day>30</day><volume>54</volume><issue>7</issue><fpage>1</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1145/3462476</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Katsis</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Rosenthal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fadnis</surname><given-names>K</given-names> </name><etal/></person-group><article-title>mt RAG: a multi-turn conversational benchmark for evaluating retrieval-augmented generation systems</article-title><source>Trans Assoc Comput Linguist</source><year>2025</year><month>07</month><day>29</day><volume>13</volume><fpage>784</fpage><lpage>808</lpage><pub-id pub-id-type="doi">10.1162/TACL.a.19</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Radlinski</surname><given-names>F</given-names> </name><name name-style="western"><surname>Craswell</surname><given-names>N</given-names> </name></person-group><article-title>Comparing the sensitivity of information retrieval metrics</article-title><conf-name>The 33rd International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR &#x2019;10)</conf-name><conf-date>Jul 19-23, 2010</conf-date><conf-loc>Geneva, Switzerland</conf-loc><fpage>667</fpage><lpage>674</lpage><pub-id pub-id-type="doi">10.1145/1835449.1835560</pub-id></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>J</given-names> </name><name name-style="western"><surname>Saha</surname><given-names>S</given-names> </name><name name-style="western"><surname>Han</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Emotional communication in HIV care: an observational study of patients&#x2019; expressed emotions and clinician response</article-title><source>AIDS Behav</source><year>2019</year><month>10</month><volume>23</volume><issue>10</issue><fpage>2816</fpage><lpage>2828</lpage><pub-id pub-id-type="doi">10.1007/s10461-019-02466-z</pub-id><pub-id pub-id-type="medline">30895426</pub-id></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="web"><article-title>Vamsi/t5_paraphrase_paws</article-title><source>Hugging Face</source><access-date>2025-03-09</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/Vamsi/T5_Paraphrase_Paws">https://huggingface.co/Vamsi/T5_Paraphrase_Paws</ext-link></comment></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abbasian</surname><given-names>M</given-names> </name><name name-style="western"><surname>Khatibi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Azimi</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Foundation metrics for evaluating effectiveness of healthcare conversations powered by generative AI</article-title><source>NPJ Digit Med</source><year>2024</year><month>03</month><day>29</day><volume>7</volume><issue>1</issue><fpage>82</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01074-z</pub-id><pub-id pub-id-type="medline">38553625</pub-id></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lam</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Li</surname><given-names>EJ</given-names> </name><etal/></person-group><article-title>Emotionally numb or empathetic? Evaluating how LLMs feel using EmotionBench</article-title><source>arXiv</source><access-date>2026-09-21</access-date><comment>Preprint posted online on  Nov 16, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2308.03656</pub-id></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Manzoor</surname><given-names>MA</given-names> </name><etal/></person-group><article-title>Factuality of large language models: a survey</article-title><access-date>2026-08-12</access-date><conf-name>2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.emnlp-main">https://aclanthology.org/2024.emnlp-main</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.1088</pub-id></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xie</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Schenck</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>F</given-names> </name></person-group><article-title>Faithful AI in medicine: a systematic review with large language models and beyond</article-title><source>medRxiv</source><year>2023</year><month>07</month><day>1</day><fpage>2023.04.18.23288752</fpage><pub-id pub-id-type="doi">10.1101/2023.04.18.23288752</pub-id><pub-id pub-id-type="medline">37398329</pub-id></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Hewett</surname><given-names>TT</given-names> </name></person-group><article-title>The role of iterative evaluation in designing systems for usability</article-title><source>Proceedings of the Second Conference of the British Computer Society, Human Computer Interaction Specialist Group on People and Computers: Designing for Usability</source><year>1986</year><publisher-name>Cambridge University Press</publisher-name><fpage>196</fpage><lpage>214</lpage><pub-id pub-id-type="doi">10.5555/17324.24085</pub-id></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A survey on evaluation of large language models</article-title><source>ACM Trans Intell Syst Technol</source><year>2024</year><month>06</month><day>30</day><volume>15</volume><issue>3</issue><fpage>1</fpage><lpage>45</lpage><pub-id pub-id-type="doi">10.1145/3641289</pub-id></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>van der Lee</surname><given-names>C</given-names> </name><name name-style="western"><surname>Gatt</surname><given-names>A</given-names> </name><name name-style="western"><surname>van Miltenburg</surname><given-names>E</given-names> </name><name name-style="western"><surname>Wubben</surname><given-names>S</given-names> </name><name name-style="western"><surname>Krahmer</surname><given-names>E</given-names> </name></person-group><article-title>Best practices for the human evaluation of automatically generated text</article-title><access-date>2026-08-12</access-date><conf-name>12th International Conference on Natural Language Generation</conf-name><conf-date>Oct 29 to Nov 1, 2019</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W19-86">https://aclanthology.org/W19-86</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/W19-8643</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Musheyev</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bockelman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Loeb</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kabarriti</surname><given-names>AE</given-names> </name></person-group><article-title>Assessment of artificial intelligence chatbot responses to top searched queries about cancer</article-title><source>JAMA Oncol</source><year>2023</year><month>10</month><day>1</day><volume>9</volume><issue>10</issue><fpage>1437</fpage><lpage>1440</lpage><pub-id pub-id-type="doi">10.1001/jamaoncol.2023.2947</pub-id><pub-id pub-id-type="medline">37615960</pub-id></nlm-citation></ref><ref id="ref97"><label>97</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Heston</surname><given-names>T</given-names> </name><name name-style="western"><surname>Khun</surname><given-names>C</given-names> </name></person-group><article-title>Prompt engineering in medical education</article-title><source>Int Med Educ</source><year>2023</year><volume>2</volume><issue>3</issue><fpage>198</fpage><lpage>205</lpage><pub-id pub-id-type="doi">10.3390/ime2030019</pub-id></nlm-citation></ref><ref id="ref98"><label>98</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>YQ</given-names> </name><name name-style="western"><surname>McCauley</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Antiretroviral therapy for the prevention of HIV-1 transmission</article-title><source>N Engl J Med</source><year>2016</year><month>09</month><day>1</day><volume>375</volume><issue>9</issue><fpage>830</fpage><lpage>839</lpage><pub-id pub-id-type="doi">10.1056/NEJMoa1600693</pub-id><pub-id pub-id-type="medline">27424812</pub-id></nlm-citation></ref><ref id="ref99"><label>99</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ayers</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Poliak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name></person-group><article-title>Responses to patient questions posted to a public social media forum</article-title><source>JAMA Intern Med</source><year>2023</year><month>04</month><day>28</day><volume>71</volume><issue>183</issue><fpage>589</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2023.1838</pub-id></nlm-citation></ref><ref id="ref100"><label>100</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>L</given-names> </name></person-group><article-title>Understanding the fundamental design decisions of retrieval-augmented generation systems</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 29, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2411.19463</pub-id></nlm-citation></ref><ref id="ref101"><label>101</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koo</surname><given-names>TK</given-names> </name><name name-style="western"><surname>Li</surname><given-names>MY</given-names> </name></person-group><article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title><source>J Chiropr Med</source><year>2016</year><month>06</month><volume>15</volume><issue>2</issue><fpage>155</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id><pub-id pub-id-type="medline">27330520</pub-id></nlm-citation></ref><ref id="ref102"><label>102</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name></person-group><article-title>ROUGE: a package for automatic evaluation of summaries</article-title><access-date>2025-03-09</access-date><conf-name>Text Summarization Branches Out (A Post-Conference Workshop of ACL 2004)</conf-name><conf-date>Jul 25, 2004</conf-date><conf-loc>Barcelona, Spain</conf-loc><fpage>74</fpage><lpage>81</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W04-1013/">https://aclanthology.org/W04-1013/</ext-link></comment></nlm-citation></ref><ref id="ref103"><label>103</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Meincke</surname><given-names>L</given-names> </name><name name-style="western"><surname>Mollick</surname><given-names>ER</given-names> </name><name name-style="western"><surname>Mollick</surname><given-names>L</given-names> </name><name name-style="western"><surname>Shapiro</surname><given-names>D</given-names> </name></person-group><article-title>Prompting science report 1: prompt engineering is complicated and contingent</article-title><year>2025</year><access-date>2026-08-10</access-date><publisher-name>Social Science Research Network</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://papers.ssrn.com/abstract=5165270">https://papers.ssrn.com/abstract=5165270</ext-link></comment><pub-id pub-id-type="doi">10.2139/ssrn.5165270</pub-id></nlm-citation></ref><ref id="ref104"><label>104</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Leidinger</surname><given-names>A</given-names> </name><name name-style="western"><surname>van Rooij</surname><given-names>R</given-names> </name><name name-style="western"><surname>Shutova</surname><given-names>E</given-names> </name></person-group><article-title>The language of prompting: what linguistic properties make a prompt successful?</article-title><access-date>2026-09-20</access-date><conf-name>Findings of the Association for Computational Linguistics: EMNLP 2023</conf-name><conf-date>Dec 6-10, 2023</conf-date><conf-loc>Singapore</conf-loc><fpage>9210</fpage><lpage>9232</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2023.findings-emnlp.618.pdf">https://aclanthology.org/2023.findings-emnlp.618.pdf</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2023.findings-emnlp.618</pub-id></nlm-citation></ref><ref id="ref105"><label>105</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lamichhane</surname><given-names>B</given-names> </name></person-group><article-title>Evaluation of chatgpt for NLP-based mental health applications</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 28, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.15727</pub-id></nlm-citation></ref><ref id="ref106"><label>106</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tay</surname><given-names>W</given-names> </name><name name-style="western"><surname>Joshi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Karimi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>S</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Mistica</surname><given-names>M</given-names> </name><name name-style="western"><surname>Piccardi</surname><given-names>M</given-names> </name><name name-style="western"><surname>MacKinlay</surname><given-names>A</given-names> </name></person-group><article-title>Red-faced ROUGH: examining the suitability of ROUGE for opinion summary evaluation</article-title><access-date>2026-09-20</access-date><conf-name>The 17th Annual Workshop of the Australasian Language Technology Association (ALTA 2019)</conf-name><conf-date>Dec 4-6, 2019</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/U19-1008/">https://aclanthology.org/U19-1008/</ext-link></comment></nlm-citation></ref><ref id="ref107"><label>107</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wallace</surname><given-names>E</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>S</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gardner</surname><given-names>M</given-names> </name></person-group><article-title>Do NLP models know numbers? probing numeracy in embeddings</article-title><access-date>2026-08-12</access-date><conf-name>2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-date>Nov 3-7, 2019</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.aclweb.org/anthology/D19-1">https://www.aclweb.org/anthology/D19-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D19-1534</pub-id></nlm-citation></ref><ref id="ref108"><label>108</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Cheng</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>van Rooij</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Z</given-names> </name></person-group><article-title>Empowering LLMs with logical reasoning: a comprehensive survey</article-title><access-date>2026-09-20</access-date><conf-name>Proceedings of the Thirty-Fourth International Joint Conference on Artificial Intelligence</conf-name><conf-date>Aug 16-22, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ijcai.org/proceedings/2025">https://www.ijcai.org/proceedings/2025</ext-link></comment><pub-id pub-id-type="doi">10.24963/ijcai.2025/1155</pub-id></nlm-citation></ref><ref id="ref109"><label>109</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zamaraeva</surname><given-names>O</given-names> </name><name name-style="western"><surname>Flickinger</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bond</surname><given-names>F</given-names> </name><name name-style="western"><surname>G&#x00F3;mez-Rodr&#x00ED;guez</surname><given-names>C</given-names> </name></person-group><article-title>Comparing LLM-generated and human-authored news text using formal syntactic theory</article-title><access-date>2026-08-12</access-date><conf-name>63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.acl-long">https://aclanthology.org/2025.acl-long</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.443</pub-id></nlm-citation></ref><ref id="ref110"><label>110</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Bulbulia</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bergen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Liut</surname><given-names>M</given-names> </name></person-group><article-title>Can small language models with retrieval-augmented generation replace large language models when learning computer science?</article-title><access-date>2026-08-12</access-date><conf-name>2024 on Innovation and Technology in Computer Science Education V 1 (ITiCSE 2024)</conf-name><conf-date>Jul 8-10, 2024</conf-date><conf-loc>Milan, Italy</conf-loc><fpage>388</fpage><lpage>393</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3649217">https://dl.acm.org/doi/proceedings/10.1145/3649217</ext-link></comment><pub-id pub-id-type="doi">10.1145/3649217.3653554</pub-id></nlm-citation></ref><ref id="ref111"><label>111</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Shandilya</surname><given-names>B</given-names> </name><name name-style="western"><surname>Palmer</surname><given-names>A</given-names> </name></person-group><article-title>Boosting the capabilities of compact models in low-data contexts with large language models and retrieval-augmented generation</article-title><access-date>2026-09-06</access-date><conf-name>31st International Conference on Computational Linguistics</conf-name><conf-date>Jan 21-24, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.coling-main.499/">https://aclanthology.org/2025.coling-main.499/</ext-link></comment></nlm-citation></ref><ref id="ref112"><label>112</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lahiri</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>QV</given-names> </name></person-group><article-title>AlzheimerRAG: multimodal retrieval-augmented generation for clinical use cases</article-title><source>Mach Learn Knowl Extr</source><year>2025</year><month>08</month><day>27</day><volume>7</volume><issue>3</issue><fpage>89</fpage><pub-id pub-id-type="doi">10.3390/make7030089</pub-id></nlm-citation></ref><ref id="ref113"><label>113</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The performance of large language model-powered chatbots compared to oncology physicians on colorectal cancer queries</article-title><source>Int J Surg</source><year>2024</year><month>10</month><day>1</day><volume>110</volume><issue>10</issue><fpage>6509</fpage><lpage>6517</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000001850</pub-id><pub-id pub-id-type="medline">38935100</pub-id></nlm-citation></ref><ref id="ref114"><label>114</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arasteh</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Lotfinia</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bressem</surname><given-names>K</given-names> </name><etal/></person-group><article-title>RadioRAG: factual large language models for enhanced diagnostics in radiology using online retrieval augmented generation</article-title><source>Radiol Artif Intell</source><year>2025</year><pub-id pub-id-type="doi">10.1148/ryai.240476</pub-id></nlm-citation></ref><ref id="ref115"><label>115</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khurana</surname><given-names>D</given-names> </name><name name-style="western"><surname>Koli</surname><given-names>A</given-names> </name><name name-style="western"><surname>Khatter</surname><given-names>K</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>S</given-names> </name></person-group><article-title>Natural language processing: state of the art, current trends and challenges</article-title><source>Multimed Tools Appl</source><year>2023</year><month>01</month><volume>82</volume><issue>3</issue><fpage>3713</fpage><lpage>3744</lpage><pub-id pub-id-type="doi">10.1007/s11042-022-13428-4s</pub-id></nlm-citation></ref><ref id="ref116"><label>116</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jabarulla</surname><given-names>MY</given-names> </name><name name-style="western"><surname>Oeltze-Jafra</surname><given-names>S</given-names> </name><name name-style="western"><surname>Beerbaum</surname><given-names>P</given-names> </name><name name-style="western"><surname>Uden</surname><given-names>T</given-names> </name></person-group><article-title>MedDoc-Bot: a chat tool for comparative analysis of large language models in the context of the pediatric hypertension guideline</article-title><access-date>2026-09-20</access-date><conf-name>2024 46th Annual International Conference of the IEEE Engineering in Medicine and Biology Society (EMBC)</conf-name><conf-date>Jul 15-19, 2024</conf-date><conf-loc>Orlando, FL, USA</conf-loc><fpage>1</fpage><lpage>4</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/document/10781509">https://ieeexplore.ieee.org/document/10781509</ext-link></comment><pub-id pub-id-type="doi">10.1109/EMBC53108.2024.10781509</pub-id></nlm-citation></ref><ref id="ref117"><label>117</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aghaziarati</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rahimi</surname><given-names>H</given-names> </name></person-group><article-title>The future of digital assistants: human dependence and behavioral change</article-title><source>J Foresight Health Gov</source><year>2025</year><access-date>2026-08-10</access-date><volume>2</volume><issue>1</issue><fpage>52</fpage><lpage>61</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://journalfhg.com/index.php/jfph/article/view/6">https://journalfhg.com/index.php/jfph/article/view/6</ext-link></comment></nlm-citation></ref><ref id="ref118"><label>118</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hipgrave</surname><given-names>L</given-names> </name><name name-style="western"><surname>Goldie</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dennis</surname><given-names>S</given-names> </name><name name-style="western"><surname>Coleman</surname><given-names>A</given-names> </name></person-group><article-title>Balancing risks and benefits: clinicians&#x2019; perspectives on the use of generative AI chatbots in mental healthcare</article-title><source>Front Digit Health</source><year>2025</year><volume>7</volume><fpage>1606291</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2025.1606291</pub-id><pub-id pub-id-type="medline">40510413</pub-id></nlm-citation></ref><ref id="ref119"><label>119</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mansurova</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mansurova</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nugumanova</surname><given-names>A</given-names> </name></person-group><article-title>QA-RAG: exploring LLM reliance on external knowledge</article-title><source>Big Data Cogn Comput</source><year>2024</year><month>09</month><volume>8</volume><issue>9</issue><fpage>115</fpage><pub-id pub-id-type="doi">10.3390/bdcc8090115</pub-id></nlm-citation></ref><ref id="ref120"><label>120</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>DH</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Design considerations for implementing eHealth behavioral interventions for HIV prevention in evolving sociotechnical landscapes</article-title><source>Curr HIV/AIDS Rep</source><year>2019</year><month>08</month><volume>16</volume><issue>4</issue><fpage>335</fpage><lpage>348</lpage><pub-id pub-id-type="doi">10.1007/s11904-019-00455-4</pub-id><pub-id pub-id-type="medline">31250195</pub-id></nlm-citation></ref><ref id="ref121"><label>121</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mohr</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Lyon</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Lattie</surname><given-names>EG</given-names> </name><name name-style="western"><surname>Reddy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schueller</surname><given-names>SM</given-names> </name></person-group><article-title>Accelerating digital mental health research from early design and creation to successful implementation and sustainment</article-title><source>J Med Internet Res</source><year>2017</year><month>05</month><volume>19</volume><issue>5</issue><fpage>e153</fpage><pub-id pub-id-type="doi">10.2196/jmir.7725</pub-id><pub-id pub-id-type="medline">28490417</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompt engineering.</p><media xlink:href="resprot_v15i1e79966_app1.docx" xlink:title="DOCX File, 21 KB"/></supplementary-material></app-group></back></article>