<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Res Protoc</journal-id><journal-id journal-id-type="publisher-id">ResProt</journal-id><journal-id journal-id-type="index">5</journal-id><journal-title>JMIR Research Protocols</journal-title><abbrev-journal-title>JMIR Res Protoc</abbrev-journal-title><issn pub-type="epub">1929-0748</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v15i1e91677</article-id><article-id pub-id-type="doi">10.2196/91677</article-id><article-categories><subj-group subj-group-type="heading"><subject>Protocol</subject></subj-group></article-categories><title-group><article-title>Frameworks, Methodologies, and Tools for Evaluating Large Language Models in Digital Mental Health Interventions: Protocol for a Scoping Review</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Salinas-Layana</surname><given-names>Antonio</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jim&#x00E9;nez-Molina</surname><given-names>&#x00C1;lvaro</given-names></name><degrees>MSc, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lira</surname><given-names>Daniela</given-names></name><degrees>MSc, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>V&#x00E9;liz-Montoya</surname><given-names>F&#x00E9;lix Alberto</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Venegas</surname><given-names>Alexi</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mu&#x00F1;oz</surname><given-names>Nicol&#x00E1;s</given-names></name><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chand&#x00ED;a</surname><given-names>Mario</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Rojas</surname><given-names>Rigoberto</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Mart&#x00ED;nez</surname><given-names>Vania</given-names></name><degrees>MSc, MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff7">7</xref></contrib></contrib-group><aff id="aff1"><institution>Doctorado en Psicoterapia, Universidad de Chile y Pontificia Universidad Cat&#x00F3;lica de Chile</institution><addr-line>Santiago</addr-line><addr-line>Regi&#x00F3;n Metropolitana</addr-line><country>Chile</country></aff><aff id="aff2"><institution>Nucleus to Improve the Mental Health of Adolescents and Youths (Imhay)</institution><addr-line>Santiago</addr-line><addr-line>Regi&#x00F3;n Metropolitana</addr-line><country>Chile</country></aff><aff id="aff3"><institution>Facultad de Psicolog&#x00ED;a y Humanidades, Universidad San Sebasti&#x00E1;n</institution><addr-line>Santiago</addr-line><addr-line>Regi&#x00F3;n Metropolitana</addr-line><country>Chile</country></aff><aff id="aff4"><institution>Center for Research and Action on Social Determination and Mental Health (CIADES)</institution><addr-line>Santiago</addr-line><addr-line>Regi&#x00F3;n Metropolitana</addr-line><country>Chile</country></aff><aff id="aff5"><institution>Departamento de Ciencias de la Educaci&#x00F3;n, &#x00C1;rea de Psicolog&#x00ED;a Social, Universidad de Burgos</institution><addr-line>Burgos, Castilla y Le&#x00F3;n</addr-line><country>Spain</country></aff><aff id="aff6"><institution>VeryMind</institution><addr-line>Santiago</addr-line><addr-line>Regi&#x00F3;n Metropolitana</addr-line><country>Chile</country></aff><aff id="aff7"><institution>Centro de Medicina Reproductiva y Desarrollo Integral del Adolescente (CEMERA), Facultad de Medicina, Universidad de Chile</institution><addr-line>Profesor Alberto Za&#x00F1;artu 1030, Independencia</addr-line><addr-line>Santiago</addr-line><addr-line>Regi&#x00F3;n Metropolitana</addr-line><country>Chile</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Izadi</surname><given-names>Reyhane</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Basmaji</surname><given-names>Timotaos</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Vania Mart&#x00ED;nez, MSc, MD, PhD, Centro de Medicina Reproductiva y Desarrollo Integral del Adolescente (CEMERA), Facultad de Medicina, Universidad de Chile, Profesor Alberto Za&#x00F1;artu 1030, Independencia, Santiago, Regi&#x00F3;n Metropolitana, 8380455, Chile, 56 229786484; <email>vmartinezn@uchile.cl</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>14</day><month>9</month><year>2026</year></pub-date><volume>15</volume><elocation-id>e91677</elocation-id><history><date date-type="received"><day>18</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>16</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>17</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Antonio Salinas-Layana, &#x00C1;lvaro Jim&#x00E9;nez-Molina, Daniela Lira, F&#x00E9;lix Alberto V&#x00E9;liz-Montoya, Alexi Venegas, Nicol&#x00E1;s Mu&#x00F1;oz, Mario Chand&#x00ED;a, Rigoberto Rojas, Vania Mart&#x00ED;nez. Originally published in JMIR Research Protocols (<ext-link ext-link-type="uri" xlink:href="https://www.researchprotocols.org">https://www.researchprotocols.org</ext-link>), 14.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Research Protocols, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.researchprotocols.org">https://www.researchprotocols.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.researchprotocols.org/2026/1/e91677"/><abstract><sec><title>Background</title><p>Digital mental health interventions (DMHIs) can help close persistent gaps in access to assessment, prevention, and treatment. Recent advances in generative AI, particularly large language models (LLMs), further expand this promise by enabling natural language understanding, personalization, and empathic interaction across assessment, support, and therapeutic contexts. However, significant evaluation challenges persist, including a lack of standardized constructs and validated instruments, which limit the comparability, reproducibility, and generalizability of the findings. No systematic synthesis currently documents the frameworks, methodologies, and tools used to evaluate LLMs in DMHIs, thereby hampering the development of a comprehensive evidence base to guide future evaluation efforts.</p></sec><sec><title>Objective</title><p>This scoping review aims to systematically map and synthesize the available evidence on frameworks, methodologies, and tools used to evaluate LLMs applied to DMHIs. Specifically, it aims to identify the constructs assessed, the instruments used, and the evaluation procedures and stages addressed.</p></sec><sec sec-type="methods"><title>Methods</title><p>Following the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) guidelines, this scoping review will search 5 electronic databases (PubMed, Scopus, Web of Science, IEEE Xplore, and ACM Digital Library) from January 1, 2019, to September 15, 2025. Eligibility criteria will encompass both empirical studies and theoretical proposals evaluating LLMs embedded within DMHIs. Studies limited to risk detection or decision support systems without an intervention component will be excluded. Data extraction will capture information on conceptual frameworks, methodological designs, evaluation procedures and tools, measured constructs, and other relevant contextual information.</p></sec><sec sec-type="results"><title>Results</title><p>The systematic search was conducted between September 1 and 15, 2025, yielding 4273 records across the 5 databases. After duplicate removal, of the 4273 records, 2980 (69.7%) remained for screening. A pilot screening exercise involving 4 independent reviewers achieved high interrater reliability (free-marginal Randolph &#x03BA;=0.81), with 76% (19/25 of the pilot sample) unanimous agreement, indicating adequate calibration of selection criteria. These figures are interim process indicators rather than final review findings as title and abstract screening of the remaining records is currently underway.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This scoping review is expected to provide one of the first systematic syntheses of frameworks, methodologies, and tools used to evaluate LLMs in DMHIs. By identifying prevailing patterns and gaps, the resulting evidence map is intended to serve as a practical reference for researchers, developers, and policymakers working toward a scientifically grounded, safe, ethical, and effective deployment of LLMs in mental health interventions.</p></sec><sec><title>Trial Registration</title><p>OSF Registries osf.io/pazyq; https://osf.io/pazyq/overview</p></sec><sec sec-type="registered-report"><title>International Registered Report Identifier (IRRID)</title><p>DERR1-10.2196/91677</p></sec></abstract><kwd-group><kwd>mental health</kwd><kwd>mental disorders</kwd><kwd>mental health services</kwd><kwd>telemedicine</kwd><kwd>therapy, computer assisted</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>digital health</kwd><kwd>chatbot</kwd><kwd>large language models</kwd><kwd>scoping review</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Improving population mental health remains a major global challenge amid the rising prevalence of mental disorders, persistent treatment gaps, and unmet demand for psychological services [<xref ref-type="bibr" rid="ref1">1</xref>]. In this context, digital mental health interventions (DMHIs) offer effective and scalable alternatives [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. However, robust real-world evidence on digital mental health technologies remains limited, with persistent gaps in rigorous evaluation frameworks, standardized methodologies, and transparency in outcome reporting [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>Recent advances in generative AI, particularly large language models (LLMs), extend this promise by enabling complex, individualized interventions at scale, surpassing earlier rule-based systems that relied on decision trees and keyword matching [<xref ref-type="bibr" rid="ref5">5</xref>]. LLMs show strong performance on &#x201C;theory-of-mind tasks&#x201D; (eg, interpreting indirect requests, tracking false beliefs, and inferring intentions), at times approaching or exceeding human levels [<xref ref-type="bibr" rid="ref6">6</xref>], and are often perceived as empathic conversational partners [<xref ref-type="bibr" rid="ref7">7</xref>]. These features enable the sensitive detection of shifts in mental state and the delivery of interactions tailored to individual needs [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Preliminary effectiveness evidence is encouraging: a recent randomized clinical trial of the therapeutic agent Therabot reported significant reductions in anxiety and depressive symptoms in a clinical population [<xref ref-type="bibr" rid="ref10">10</xref>]. However, the overall strength and reliability of this emerging evidence base warrant cautious interpretation as many recent AI chatbot trials for anxiety and depression are characterized by small and demographically narrow samples; suboptimal or inappropriate control conditions; and heterogeneous, nonstandardized outcome measures that limit both the generalizability of their findings and the reproducibility of their procedures [<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>As LLMs are increasingly integrated into DMHIs, it is essential to understand and systematize their evaluation process. This imperative is underscored by ethical challenges such as responsiveness to risk scenarios [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. The scoping review by Hua et al [<xref ref-type="bibr" rid="ref8">8</xref>] identified substantive limitations such as the lack of standardized constructs and scales and the resulting proliferation of ad hoc instruments with uncertain validity and reliability, which hampers cross-study comparisons and weakens the evidence base. The literature also overemphasizes user experience constructs (eg, usability, accessibility, and perceived usefulness) at the expense of critical dimensions such as safety, privacy, and equity. In contrast, other constructs such as transparency, resilience, accountability, and explainability remain largely unexplored. Complementing this, the scoping review by Jin et al [<xref ref-type="bibr" rid="ref14">14</xref>] on LLM applications in mental health systematized the most commonly used performance metrics (<italic>F</italic><sub>1</sub>-score, precision, accuracy, and recall) to evaluate interactions both among LLMs and between LLMs and human professionals. While these metrics provide a quantitative foundation, the analysis highlighted important gaps: the limitations of LLMs in addressing complex clinical tasks and the need to balance operational efficiency with potential risks. The aforementioned review also underscored the risk-benefit trade-off, highlighting 3 sensitive domains: data privacy, model bias, and the ethical implications of clinical implementation [<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>While these recent scoping reviews have provided foundational insights into the use of LLMs in mental health, they remain insufficient to address the specific objectives of the present study. The review by Hua et al [<xref ref-type="bibr" rid="ref8">8</xref>] primarily categorizes what evaluation constructs are measured across broad generative applications (including clinical assistants and diagnostic aids for health care providers) rather than systematically examining the conceptual frameworks, evaluation methodologies, and specific tools used exclusively within direct DMHIs. Similarly, the focus by Jin et al [<xref ref-type="bibr" rid="ref14">14</xref>] on computational task performance metrics (eg, <italic>F</italic><sub>1</sub>-score and precision) is insufficient to capture the comprehensive methodological approaches required to evaluate clinical interventions, which must also account for therapeutic appropriateness, clinical safety, and user-centered outcomes through structured clinical tools.</p><p>Therefore, to the best of our knowledge, no systematic synthesis currently documents the conceptual frameworks, methodologies, and tools used or proposed to evaluate LLMs within mental health interventions. This hampers understanding of the field, the identification of best practices, and the detection of evidence gaps, thereby constraining the development of rigorous and evidence-informed evaluation approaches. A scoping review is therefore an appropriate approach as it maps the breadth, nature, and characteristics of the available literature, including empirical studies and theoretical proposals, and provides an integrated overview of the state of the field [<xref ref-type="bibr" rid="ref15">15</xref>]. Unlike effectiveness-focused systematic reviews, scoping reviews are well suited to emerging, methodologically heterogeneous areas. While they do not involve formal critical appraisal, they record reported limitations during data extraction [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>In this way, this protocol structures a scoping review to answer the following question: what evidence exists regarding the frameworks, methodologies, and tools used to evaluate LLMs applied to DMHIs?</p><p>The following complementary questions are posed: (1) what constructs have been measured and with which instruments? (2) Which evaluation phases have been addressed?</p><p>The objective of the scoping review is, therefore, to systematize the evidence regarding the evaluation processes of LLMs aimed at DMHIs, identifying the frameworks, methodologies, and tools used.</p><p>For this purpose, a framework will be understood as a conceptual structure that organizes and defines which dimensions to evaluate and how they are related without specifying a particular procedure for carrying out the evaluation [<xref ref-type="bibr" rid="ref17">17</xref>]. Methodology will be understood as the procedures carried out to conduct the evaluation [<xref ref-type="bibr" rid="ref18">18</xref>]. For this review, methodology will include 3 core components: study design, measured constructs, and the type of analysis. Finally, a tool will be understood as a concrete resource used in evaluation, such as questionnaires, interviews, and chat session transcripts, among others.</p><p>For this protocol, DMHIs are defined as programs, applications, chatbots, virtual agents, or digital platforms designed to deliver therapeutic, preventive, or psychosocial support components through digital technologies with or without direct clinical supervision. Accordingly, DMHIs include applications, chatbots, virtual agents, or platforms that provide mental health intervention, but tools intended exclusively for decision support, data analysis, or risk identification without an intervention component are excluded. LLMs are understood as computational AI models trained on large volumes of textual data that use deep architectures (such as transformers) and contain billions of parameters, enabling them to identify complex language patterns and generate coherent responses to a wide variety of natural language processing tasks [<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>To define and identify the evaluation phases, the framework proposed by Ding et al [<xref ref-type="bibr" rid="ref20">20</xref>] for conversational agents with AI in health interventions will be used. This framework proposes 4 sequential stages aligned with progression from initial testing to large-scale implementation: feasibility and usability, efficacy, effectiveness, and implementation.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Sources of Information</title><p>The PubMed, Scopus, Web of Science, IEEE Xplore, and ACM Digital Library databases will be consulted to ensure a comprehensive and balanced coverage of biomedical, multidisciplinary, and technological research. The search period will span January 1, 2019, to September 15, 2025. This time frame was selected because the transformer architecture, which underpins LLMs, was introduced in 2017 [<xref ref-type="bibr" rid="ref21">21</xref>], and GPT-2, one of the first widely available LLMs, was released in 2019. The latter marked a turning point by enabling the widespread development and application of LLMs in various domains, including digital health [<xref ref-type="bibr" rid="ref22">22</xref>].</p><p>To capture emerging developments and early-stage proposals, preprints indexed in the Web of Science Preprint Citation Index will be included. All search activities will be recorded and documented to ensure transparency and reproducibility.</p></sec><sec id="s2-2"><title>Eligibility Criteria</title><p>The inclusion and exclusion criteria were defined to ensure the relevance of the studies identified in the review. The population, concept, and context framework recommended by the Joanna Briggs Institute (JBI) [<xref ref-type="bibr" rid="ref16">16</xref>] will be used as a guiding criterion primarily focusing on the concept and context domains given that, based on the objective of this scoping review, the population category is not informative. The concept refers to the frameworks, methodologies, and tools used to evaluate LLMs in DMHIs. The context encompasses laboratory and real-world scenarios as well as theoretical assessment proposals without establishing geographical or cultural restrictions to broadly capture the diversity of approaches.</p><p>Research published in recognized databases and recent gray literature will be included without language restrictions to capture both established and emerging production. On the other hand, works with a general focus on AI without explicit reference to LLMs, those in which the models do not incorporate a mental health intervention component, applications limited to data analysis or administrative tasks, and literature reviews will be excluded. Details on the criteria are provided in <xref ref-type="other" rid="box1">Textbox 1</xref>.</p><boxed-text id="box1"><title> Inclusion and exclusion criteria.</title><p><bold>Inclusion criteria</bold></p><list list-type="bullet"><list-item><p>Empirical studies that evaluate large language models (LLMs) in the context of digital mental health interventions (DMHIs)</p></list-item><list-item><p>Theoretical proposals for evaluating LLMs in DMHIs, including conceptual frameworks, methodological approaches, or evaluation tools</p></list-item><list-item><p>Studies published in the PubMed, Scopus, Web of Science, IEEE Xplore, and ACM Digital Library databases from January 1, 2019, to September 15, 2025</p></list-item><list-item><p>Preprint gray literature studies available in the Web of Science Preprint Citation Index</p></list-item><list-item><p>Studies published in any language if an English-language abstract is available for screening</p></list-item></list><p><bold>Exclusion criteria</bold></p><list list-type="bullet"><list-item><p>Studies that use broad terms such as &#x201C;AI&#x201D; or &#x201C;generative AI&#x201D; but do not specifically refer to the use of an LLM in the intervention</p></list-item><list-item><p>Studies in which the LLM does not incorporate a specific component intended for mental health interventions (a specific component is understood to mean, eg, providing emotional support, generating dialogue for therapeutic purposes, or providing behavior change strategies focused on mental health)</p></list-item><list-item><p>Studies in which the LLM is used for data analysis, diagnosis, or administrative support without an intervention component</p></list-item><list-item><p>Studies that are literature reviews, such as systematic reviews or scoping reviews</p></list-item></list></boxed-text><p>To improve screening consistency, eligibility criteria will be operationalized using 2 complementary decision rules applied during study selection. First, studies will be required to explicitly report or propose methods for evaluating LLM-based systems, including empirical evaluations using predefined criteria (eg, safety, usability, and clinical relevance) or theoretical contributions such as structured frameworks, methodologies, or tools for assessment. Studies lacking an explicit evaluative component will be excluded. Second, studies will be required to involve LLM-based systems within the DMHI context as defined in the population, concept, and context framework. Systems without an evaluative focus or without an intervention-oriented function will be excluded.</p><p>To further delimit boundary cases, the following worked examples operationalize the eligibility criteria:</p><list list-type="bullet"><list-item><p>Included&#x2014;an LLM-based chatbot delivering therapeutic dialogue, emotional support, or behavior change strategies that are evaluated against defined criteria (eg, safety, usability, or clinical relevance) and a conceptual framework, methodology, or tool proposed to evaluate LLM-based DMHIs</p></list-item><list-item><p>Excluded&#x2014;a chatbot described only at the design or architecture level without any evaluation, an LLM used solely for suicide risk detection or triage without support or an intervention delivered to the user, an LLM clinical decision support tool assisting clinicians rather than delivering an intervention to end users, and a study referring to &#x201C;AI&#x201D; or &#x201C;generative AI&#x201D; without specifying the use of an LLM in the intervention</p></list-item></list><p>Ambiguous or borderline cases will be resolved through discussion among reviewers, with arbitration by a senior reviewer when consensus is not reached.</p></sec><sec id="s2-3"><title>Search Strategy</title><p>The search strategy was designed in a replicable manner through an iterative process within the research team based on previous exploratory searches conducted by the team and a published protocol [<xref ref-type="bibr" rid="ref23">23</xref>]. An initial set of key terms and Boolean operators was developed for each database in relation to three main concepts: (1) assessment, metrics, or frameworks; (2) LLMs; and (3) DMHIs. After an initial search process, the &#x201C;NOT&#x201D; operator was incorporated to reduce irrelevant records (eg, unrelated health conditions and review articles) following iterative testing conducted while minimizing the risk of excluding relevant studies within the scope of the review.</p><p>The complete strategy used for Web of Science can be found in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. It served as a template for adapting searches across the other platforms, with the necessary modifications applied to maximize the identification of relevant studies. The search was conducted in all predefined databases without language restrictions. The period was restricted to publications from January 1, 2019, to September 15, 2025. Gray literature was searched through the Web of Science Preprint Citation Index and supplemented by identifying studies in the reference lists of included articles and previous reviews.</p></sec><sec id="s2-4"><title>Article Selection</title><p>The studies selected in the search will be reviewed to eliminate duplicates using 2 software programs. First, EndNote 2025 (Clarivate Analytics) will be used for an initial reference import and to remove duplicates. Subsequently, the data will be exported to Rayyan (Qatar Computing Research Institute), where a second duplicate check will be performed.</p><p>The study screening process will be carried out in 2 phases using the Rayyan platform. Before the formal screening process, a pilot title and abstract screening exercise was conducted on a random sample of 25 records to assess the feasibility of the eligibility criteria and calibrate reviewer agreement. Four independent reviewers applied the predefined inclusion and exclusion criteria, classifying each record as included, excluded, or potentially eligible. This calibration process was repeated until an agreement level of 75% or higher was achieved, consistent with previously established standards [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>Following this pilot phase, the formal screening process will begin. In the first phase, title and abstract screening will be conducted using the full dataset. The remaining records will be divided into 3 blocks of equal size. One reviewer will screen all blocks, whereas the other 3 reviewers will independently screen 1 block each. All records will undergo double screening at the title and abstract level, with disagreements resolved through consensus with a senior researcher.</p><p>In the second phase, the full texts of the preselected studies will be evaluated for inclusion by pairs of reviewers consisting of a mental health professional and a technology professional. Disagreements will initially be resolved through discussion between the 2 reviewers. In cases in which consensus cannot be reached, a third evaluator, selected according to the nature of the disagreement (mental health or technology), will intervene and make the final decision. This procedure aims to ensure the reliability of the selection process and integrate complementary disciplinary perspectives that help minimize potential biases. Given the emerging maturity of the field and the objective of this review, a quality assessment of the articles will not be conducted. Articles in languages other than English or Spanish will be translated using GPT-5 (OpenAI).</p><p>The entire process of identification, screening, inclusion, and exclusion will be described narratively and illustrated using a flowchart in accordance with the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) recommendations in the final manuscript (<xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>).</p></sec><sec id="s2-5"><title>Data Extraction</title><p>A data charting form developed by the research team will be used, informed by the objectives of this scoping review, the JBI guidance for scoping reviews [<xref ref-type="bibr" rid="ref16">16</xref>], and preliminary pilot-testing. The charting framework is structured to capture four main categories relevant to the review questions: (1) study and intervention characteristics, including publication details, population, and DMHI purpose; (2) LLM characteristics, including model name and version and accessibility characteristics; (3) evaluation characteristics, including reported evaluation frameworks, evaluation phase, evaluation target, evaluation approach and methodology, data source and sample, evaluation domains, measured constructs, indicators and metrics, assessment instruments or tools, and evaluation findings and outcomes; and (4) study limitations reported by the authors.</p><p>The charting framework will include a codebook providing operational definitions for each data extraction field, with examples where applicable. The codebook will specify that only information explicitly reported in the included studies will be extracted and that fields will be coded as &#x201C;not reported&#x201D; when relevant information is unavailable.</p><p>To improve consistency in the extraction of evaluation-related information, particularly given the expected heterogeneity in terminology across studies, an a priori set of evaluation domains and measured constructs informed by the existing literature [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref20">20</xref>] will be used as a procedural tool to guide data charting and standardize the identification and coding of reported assessment areas. This set is not intended as a fixed taxonomy or comprehensive classification system but rather as a pragmatic starting structure that supports consistent extraction and synthesis across studies. The set will remain iterative, allowing for the refinement of domains and constructs and the incorporation of additional categories identified inductively during the charting process.</p><p>Evaluation domains will provide a high-level structure for organizing and synthesizing broader areas of assessment reported in the literature, whereas measured constructs will capture the specific attributes, processes, or outcomes evaluated within each domain. The initial set of evaluation domains includes safety and risk response, clinical efficacy and effectiveness, usability, user experience and engagement, therapeutic interaction and relational quality, explainability and transparency, bias, fairness and equity, privacy and data governance, accountability and clinical governance, and technical performance and reliability. These domains reflect key dimensions of evaluation that are both commonly assessed and increasingly highlighted in the literature on AI-based interventions in health care, including areas that remain underrepresented but are considered important for comprehensive assessment. Each domain is defined in the codebook and paired with example constructs and indicators, as detailed in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. All modifications will be documented, including their rationale and timing, and will be reported in the final review manuscript.</p><p>The extraction will be conducted by 6 reviewers organized into 3 interdisciplinary pairs, with each pair consisting of one mental health expert and one technology expert. The process will unfold in 2 stages.</p><p>In the first stage, 3 studies will be randomly selected for pilot extraction. Each interdisciplinary pair will independently extract data from these studies, working collaboratively in real time using shared extraction forms. Following independent extraction by all 3 pairs, the whole team will convene to calibrate criteria, resolve inconsistencies, and ensure interdisciplinary alignment before proceeding to the main extraction phase.</p><p>In the second stage (main extraction), after calibration, the remaining studies will be divided into 3 equal blocks, with each block assigned to 1 of the 3 interdisciplinary pairs. Pairs will work collaboratively using shared extraction forms, with each expert completing domain-specific fields while maintaining ongoing dialogue to contextualize findings. To enhance accuracy and verification, the extraction process will incorporate human-AI collaboration [<xref ref-type="bibr" rid="ref24">24</xref>] using NotebookLM (Google). NotebookLM will be used to support cross-checking of extracted information, verification of technical and clinical terminology, and identification of potential inconsistencies. However, all extracted data will undergo final human validation to ensure accuracy and interpretive rigor.</p><p>Discrepancies in data extraction will be resolved through discussion between the pair of reviewers. If a consensus cannot be reached, a third senior researcher will serve as the arbitrator. The software Zotero (Corporation for Digital Scholarship) will be used for bibliographic management, and the extraction process will be conducted in Google Workspace, facilitating synchronous work, change traceability, and transparent recording of modifications.</p></sec><sec id="s2-6"><title>Data Analysis and Presentation of Results</title><p>The analysis will follow a descriptive and thematic approach aimed at mapping frameworks, methodologies, and tools used for evaluating LLMs in DMHIs. In accordance with the guidelines for scoping reviews [<xref ref-type="bibr" rid="ref16">16</xref>], the aim will not be to synthesize or assess the certainty of the results but rather to organize the evidence in a structured manner.</p><p>The synthesis will be based on the grouping of extracted data into predefined categories aligned with the research questions. Studies will be grouped according to evaluation frameworks, methodologies, and tools used, as well as by the constructs and instruments reported within each evaluation domain. In addition, evaluation phases will be used as an organizing category to describe how and when evaluation is conducted across studies.</p><p>Within each category, results will be summarized descriptively to identify how frequently specific frameworks, constructs, instruments, and phases are reported and how they are distributed across the literature. This will allow for a structured comparison of evaluation approaches without attempting to generate inferential or effect-based conclusions.</p><p>The findings will be presented using tables; narrative summaries; and, where relevant, diagrams or concept maps illustrating patterns and relationships. The presentation will follow an inductive approach and will be reported in accordance with the PRISMA-ScR guidelines.</p></sec><sec id="s2-7"><title>Protocol Registration</title><p>This scoping review protocol was registered in the Open Science Framework on November 10, 2025 (digital object identifier: 10.17605/OSF.IO/PAZYQ). The systematic database search was conducted between September 1 and 15, 2025, and a pilot title and abstract screening exercise was completed before protocol registration. These preliminary activities were conducted to assess the feasibility of the search strategy and refine the screening procedures. Registration was completed before the initiation of the formal screening phase and before any full-text assessment, data extraction, or evidence synthesis activities. The registration record includes the complete search strategy, eligibility criteria, screening and extraction procedures, and data extraction framework. Any amendments to the protocol will be documented with rationale and date in the Open Science Framework registration and transparently reported in the final manuscript in accordance with PRISMA-ScR and <italic>JBI Manual for Evidence Synthesis</italic> guidelines.</p></sec><sec id="s2-8"><title>Ethical Considerations</title><p>As this is a scoping review, participant recruitment does not apply, and ethics approval is not required.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>The following results represent interim process indicators from the screening phase of the review. Article selection is still ongoing, and these findings should not be interpreted as final review outcomes.</p><p>The study was partially supported by research funding, a doctoral scholarship supporting the principal investigator&#x2019;s doctoral training, and institutional support for a coauthor. Therefore, no single specific funding date applies to the study as a whole. Study conceptualization began in August 2025, and the systematic database search was conducted from September 1 to 15, 2025, across 5 bibliographic databases. At the time of submission, the review was in the data extraction phase, and data analysis had not yet begun. A total of 4273 records were identified from the following databases: 1597 (37.4%) from Web of Science (Core Collection and Preprint Citation Index), 830 (19.4%) from PubMed, 1329 (31.1%) from Scopus, 386 (9.0%) from IEEE Xplore, and 131 (3.1%) from ACM Digital Library. Of the 4273 records, after removing 1293 (30.3%) duplicates, 2980 (69.7%) remained for title and abstract screening (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Preliminary PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) flowchart.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e91677_fig01.png"/></fig><p>A pilot selection test was carried out based on the screening of titles and abstracts. High interrater agreement was observed among evaluators, with a free-marginal Randolph &#x03BA; coefficient [<xref ref-type="bibr" rid="ref25">25</xref>] of 0.81 calculated in 3 nominal categories (0=no, 1=yes, and 2=maybe), indicating almost perfect agreement beyond chance. This coefficient was selected for this phase because it does not impose fixed marginal distributions and is appropriate when the prevalence of categories may be unbalanced among reviewers. Complementary descriptive metrics showed unanimous agreement of 76% (19/25 records in the pilot sample for which all 4 reviewers assigned the same eligibility category [included, excluded, or potentially eligible]) and mean pairwise agreement of 87.3% (131/150 pairwise record comparisons, SD 4.68%), supporting the procedural consistency of the selection criteria and thereby supporting the initiation of formal study selection.</p><p>The final results are expected to be prepared and submitted for publication in December 2026.</p></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Expected Findings</title><p>We anticipate that this review will provide a structured overview of the frameworks, methodologies, and tools used to evaluate LLMs in DMHIs, with particular attention to the constructs, instruments, and evaluation phases reported across studies. We expect to find a field still concentrated in early-stage feasibility and usability evaluations characterized by a predominance of ad hoc instruments over validated clinical measures; uneven coverage of the safety, privacy, equity, and accountability domains; and only emerging contributions addressing the more advanced efficacy, effectiveness, and implementation stages. A scoping review is the most appropriate methodological approach given the nascent and heterogeneous nature of the field, characterized by conceptual diversity and rapid technological expansion [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref26">26</xref>].</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>Existing reviews have begun to examine LLMs in mental health contexts. Hua et al [<xref ref-type="bibr" rid="ref8">8</xref>] identified critical inconsistencies in evaluation methodologies, finding that most studies rely on ad hoc scales rather than validated clinical instruments and that safety, privacy, and algorithmic accountability remain largely unaddressed in the literature. Similarly, Jin et al [<xref ref-type="bibr" rid="ref14">14</xref>] highlighted that LLMs should function as complementary tools rather than replacements for human clinicians and emphasized the absence of rigorous ethical and safety evaluation standards. However, these reviews did not primarily focus on mapping the evaluation frameworks and methodologies used to assess LLM-based DMHIs in diverse contexts. This scoping review aims to address this gap by systematically charting existing approaches and identifying areas where further methodological refinement may be needed.</p></sec><sec id="s4-3"><title>Strengths and Limitations</title><p>The protocol presents several methodological strengths, including strict adherence to PRISMA-ScR and JBI guidelines [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]; a comprehensive search strategy that spans biomedical, technological, and multidisciplinary databases; and the inclusion of gray literature to capture emerging developments. A key limitation is the potential incomplete capture of unindexed or non&#x2013;English-language literature, which may affect the comprehensiveness of the evidence map. Additionally, given the rapid evolution of LLM technologies, some of the evidence, evaluation approaches, or practices identified in this review may change over time, which should be considered when interpreting the long-term relevance and applicability of the findings.</p></sec><sec id="s4-4"><title>Future Directions</title><p>By systematically mapping current evaluation practices, this review will provide an overview of how LLM-based DMHIs are currently being assessed and highlight areas where further research is needed. The findings may inform future work aimed at developing more comprehensive evaluation approaches that better capture the multifaceted and context-dependent nature of human-LLM interactions in mental health settings, including clinical, ethical, and user-centered dimensions.</p><p>Results will be disseminated through peer-reviewed publication, conference presentations, and a plain-language summary to facilitate access among relevant stakeholders.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This scoping review is expected to provide one of the first systematic evidence syntheses of the frameworks, methodologies, and tools used to evaluate LLMs in DMHIs. By identifying prevailing patterns and gaps, the resulting evidence map is intended to serve as a practical reference for researchers, developers, and policymakers working toward a scientifically grounded, safe, ethical, and effective deployment of LLMs in mental health interventions.</p></sec></sec></body><back><ack><p>The authors wish to acknowledge the use of a generative AI tool during the preparation of this manuscript. Specifically, NotebookLM (Google) was used to support proofreading and translation tasks. All outputs generated by this tool were reviewed, verified, and revised by the authors, who bear full responsibility for the accuracy, integrity, and content of the final manuscript. This tool was not listed as an author and assumes no authorship responsibility.</p></ack><notes><sec><title>Funding</title><p>This work was partially supported by the National Agency for Research and Development (Chile) through a National Doctoral Scholarship 2025 (grant 21250516) and the FONDECYT REGULAR Grant No. 1262084. AJ-M receives support from the Center for Research and Action on Social Determination and Mental Health (CIADES), funded by the National Centers of Interest initiative of the National Agency for Research and Development (grant CIADES CIN250054). The funders had no role in the design of the protocol; collection, analysis, and interpretation of the data; writing of the manuscript; or decision to submit it for publication.</p></sec><sec><title>Data Availability</title><p>Data sharing is not applicable to this article as no data sets were generated or analyzed during this study.</p></sec></notes><fn-group><fn fn-type="con"><p>AS-L and VM contributed to conceptualization. AS-L, DL, and FAV-M contributed to methodology (eligibility criteria, data charting framework, and analysis plan). AS-L, AV, NM, MC, RR, and AJ-M contributed to search strategy development. AS-L, DL, FAV-M, AJ-M, and NM contributed to pilot-testing of screening and data extraction forms. AS-L, AJ-M, and VM contributed to writing&#x2014;original draft. AS-L, VM, AJ-M, DL, FAV-M, AV, NM, MC, and RR contributed to writing&#x2014;review and editing.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">DMHI</term><def><p>digital mental health intervention</p></def></def-item><def-item><term id="abb2">JBI</term><def><p>Joanna Briggs Institute</p></def></def-item><def-item><term id="abb3">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb4">PRISMA-ScR</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>V</given-names> </name><name name-style="western"><surname>Saxena</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lund</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The Lancet Commission on global mental health and sustainable development</article-title><source>Lancet</source><year>2018</year><month>10</month><day>27</day><volume>392</volume><issue>10157</issue><fpage>1553</fpage><lpage>1598</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(18)31612-X</pub-id><pub-id pub-id-type="medline">30314863</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jim&#x00E9;nez-Molina</surname><given-names>&#x00C1;</given-names> </name><name name-style="western"><surname>Franco</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mart&#x00ED;nez</surname><given-names>V</given-names> </name><name name-style="western"><surname>Mart&#x00ED;nez</surname><given-names>P</given-names> </name><name name-style="western"><surname>Rojas</surname><given-names>G</given-names> </name><name name-style="western"><surname>Araya</surname><given-names>R</given-names> </name></person-group><article-title>Internet-based interventions for the prevention and treatment of mental disorders in Latin America: a scoping review</article-title><source>Front Psychiatry</source><year>2019</year><volume>10</volume><fpage>664</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2019.00664</pub-id><pub-id pub-id-type="medline">31572242</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Philippe</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Sikder</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jackson</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Digital health interventions for delivery of mental health care: systematic and comprehensive meta-review</article-title><source>JMIR Ment Health</source><year>2022</year><month>05</month><day>12</day><volume>9</volume><issue>5</issue><fpage>e35159</fpage><pub-id pub-id-type="doi">10.2196/35159</pub-id><pub-id pub-id-type="medline">35551058</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name><name name-style="western"><surname>Linardon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Goldberg</surname><given-names>SB</given-names> </name><etal/></person-group><article-title>The evolving field of digital mental health: current evidence and implementation issues for smartphone apps, generative artificial intelligence, and virtual reality</article-title><source>World Psychiatry</source><year>2025</year><month>06</month><volume>24</volume><issue>2</issue><fpage>156</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.1002/wps.21299</pub-id><pub-id pub-id-type="medline">40371757</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stade</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Eichstaedt</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Stirman</surname><given-names>SW</given-names> </name></person-group><article-title>Readiness Evaluation for AI-Mental Health Deployment and Implementation (READI): a review and proposed framework</article-title><source>Technol Mind Behav</source><year>2025</year><volume>6</volume><issue>2</issue><pub-id pub-id-type="doi">10.1037/tmb0000163</pub-id><pub-id pub-id-type="medline">41048253</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Strachan</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Albergo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Borghini</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Testing theory of mind in large language models and humans</article-title><source>Nat Hum Behav</source><year>2024</year><month>07</month><volume>8</volume><issue>7</issue><fpage>1285</fpage><lpage>1295</lpage><pub-id pub-id-type="doi">10.1038/s41562-024-01882-z</pub-id><pub-id pub-id-type="medline">38769463</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Campellone</surname><given-names>TR</given-names> </name><name name-style="western"><surname>Flom</surname><given-names>M</given-names> </name><name name-style="western"><surname>Montgomery</surname><given-names>RM</given-names> </name><etal/></person-group><article-title>Safety and user experience of a generative artificial intelligence digital mental health intervention: exploratory randomized controlled trial</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>23</day><volume>27</volume><fpage>e67365</fpage><pub-id pub-id-type="doi">10.2196/67365</pub-id><pub-id pub-id-type="medline">40408143</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hua</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Na</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A scoping review of large language models for generative tasks in mental health care</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>30</day><volume>8</volume><issue>1</issue><fpage>230</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01611-4</pub-id><pub-id pub-id-type="medline">40307331</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Malgaroli</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schultebraucks</surname><given-names>K</given-names> </name><name name-style="western"><surname>Myrick</surname><given-names>KJ</given-names> </name><etal/></person-group><article-title>Large language models for the mental health community: framework for translating code to care</article-title><source>Lancet Digit Health</source><year>2025</year><month>04</month><volume>7</volume><issue>4</issue><fpage>e282</fpage><lpage>e285</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(24)00255-3</pub-id><pub-id pub-id-type="medline">39779452</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Heinz</surname><given-names>MV</given-names> </name><name name-style="western"><surname>Mackin</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Trudeau</surname><given-names>BM</given-names> </name><etal/></person-group><article-title>Randomized trial of a generative AI chatbot for mental health treatment</article-title><source>NEJM AI</source><year>2025</year><month>03</month><day>27</day><volume>2</volume><issue>4</issue><pub-id pub-id-type="doi">10.1056/AIoa2400802</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bodner</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>K</given-names> </name><name name-style="western"><surname>Schneider</surname><given-names>R</given-names> </name><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name></person-group><article-title>Efficacy and risks of artificial intelligence chatbots for anxiety and depression: a narrative review of recent clinical studies</article-title><source>Curr Opin Psychiatry</source><year>2026</year><month>01</month><day>1</day><volume>39</volume><issue>1</issue><fpage>19</fpage><lpage>25</lpage><pub-id pub-id-type="doi">10.1097/YCO.0000000000001048</pub-id><pub-id pub-id-type="medline">41198140</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>De Freitas</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>IG</given-names> </name></person-group><article-title>The health risks of generative AI-based wellness apps</article-title><source>Nat Med</source><year>2024</year><month>05</month><volume>30</volume><issue>5</issue><fpage>1269</fpage><lpage>1275</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-02943-6</pub-id><pub-id pub-id-type="medline">38684859</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sarkar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gaur</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>LK</given-names> </name><name name-style="western"><surname>Garg</surname><given-names>M</given-names> </name><name name-style="western"><surname>Srivastava</surname><given-names>B</given-names> </name></person-group><article-title>A review of the explainability and safety of conversational agents for mental health to identify avenues for improvement</article-title><source>Front Artif Intell</source><year>2023</year><volume>6</volume><fpage>1229805</fpage><pub-id pub-id-type="doi">10.3389/frai.2023.1229805</pub-id><pub-id pub-id-type="medline">37899961</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>P</given-names> </name><etal/></person-group><article-title>The applications of large language models in mental health: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>5</day><volume>27</volume><fpage>e69284</fpage><pub-id pub-id-type="doi">10.2196/69284</pub-id><pub-id pub-id-type="medline">40324177</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Lillie</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zarin</surname><given-names>W</given-names> </name><etal/></person-group><article-title>PRISMA extension for Scoping Reviews (PRISMA-ScR): checklist and explanation</article-title><source>Ann Intern Med</source><year>2018</year><month>10</month><day>2</day><volume>169</volume><issue>7</issue><fpage>467</fpage><lpage>473</lpage><pub-id pub-id-type="doi">10.7326/M18-0850</pub-id><pub-id pub-id-type="medline">30178033</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Peters</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Godfrey</surname><given-names>C</given-names> </name><name name-style="western"><surname>McInerney</surname><given-names>P</given-names> </name><name name-style="western"><surname>Munn</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Khalil</surname><given-names>H</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Aromataris</surname><given-names>E</given-names> </name><name name-style="western"><surname>Lockwood</surname><given-names>C</given-names> </name><name name-style="western"><surname>Porritt</surname><given-names>K</given-names> </name><name name-style="western"><surname>Pilla</surname><given-names>B</given-names> </name><name name-style="western"><surname>Jordan</surname><given-names>Z</given-names> </name></person-group><article-title>Scoping reviews</article-title><source>JBI Manual for Evidence Synthesis</source><year>2024</year><publisher-name>JBI</publisher-name><pub-id pub-id-type="doi">10.46658/JBIMES-24-09</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Partelow</surname><given-names>S</given-names> </name></person-group><article-title>What is a framework? Understanding their purpose, value, development and use</article-title><source>J Environ Stud Sci</source><year>2023</year><month>09</month><volume>13</volume><fpage>510</fpage><lpage>519</lpage><pub-id pub-id-type="doi">10.1007/s13412-023-00833-w</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fortino</surname><given-names>G</given-names> </name><name name-style="western"><surname>Savaglio</surname><given-names>C</given-names> </name><name name-style="western"><surname>Spezzano</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>M</given-names> </name></person-group><article-title>Internet of things as system of systems: a review of methodologies, frameworks, platforms, and tools</article-title><source>IEEE Trans Syst Man Cybern Syst</source><year>2021</year><volume>51</volume><issue>1</issue><fpage>223</fpage><lpage>236</lpage><pub-id pub-id-type="doi">10.1109/TSMC.2020.3042898</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raiaan</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Mukta</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Fatema</surname><given-names>K</given-names> </name><etal/></person-group><article-title>A review on large language models: architectures, applications, taxonomies, open issues and challenges</article-title><source>IEEE Access</source><year>2024</year><volume>12</volume><fpage>26839</fpage><lpage>26874</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2024.3365742</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ding</surname><given-names>H</given-names> </name><name name-style="western"><surname>Simmich</surname><given-names>J</given-names> </name><name name-style="western"><surname>Vaezipour</surname><given-names>A</given-names> </name><name name-style="western"><surname>Andrews</surname><given-names>N</given-names> </name><name name-style="western"><surname>Russell</surname><given-names>T</given-names> </name></person-group><article-title>Evaluation framework for conversational agents with artificial intelligence in health interventions: a systematic scoping review</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>02</month><day>16</day><volume>31</volume><issue>3</issue><fpage>746</fpage><lpage>761</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad222</pub-id><pub-id pub-id-type="medline">38070173</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Attention is all you need</article-title><source>NIPS&#x2019;17: Proceedings of the 31st International Conference on Neural Information Processing Systems</source><year>2017</year><publisher-name>Curran Associates Inc</publisher-name><fpage>6000</fpage><lpage>6010</lpage><pub-id pub-id-type="doi">10.5555/3295222.3295349</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Revolutionizing health care: the transformative impact of large language models in medicine</article-title><source>J Med Internet Res</source><year>2025</year><month>01</month><day>7</day><volume>27</volume><fpage>e59069</fpage><pub-id pub-id-type="doi">10.2196/59069</pub-id><pub-id pub-id-type="medline">39773666</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gautam</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kellmeyer</surname><given-names>P</given-names> </name></person-group><article-title>Exploring the credibility of large language models for mental health support: protocol for a scoping review</article-title><source>JMIR Res Protoc</source><year>2025</year><month>01</month><day>29</day><volume>14</volume><fpage>e62865</fpage><pub-id pub-id-type="doi">10.2196/62865</pub-id><pub-id pub-id-type="medline">39879615</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tomczyk</surname><given-names>P</given-names> </name><name name-style="western"><surname>Br&#x00FC;ggemann</surname><given-names>P</given-names> </name><name name-style="western"><surname>Vrontis</surname><given-names>D</given-names> </name></person-group><article-title>AI meets academia: transforming systematic literature reviews</article-title><source>EuroMed J Bus</source><year>2026</year><month>03</month><day>6</day><volume>21</volume><issue>1</issue><fpage>345</fpage><lpage>369</lpage><pub-id pub-id-type="doi">10.1108/EMJB-03-2024-0055</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Randolph</surname><given-names>JJ</given-names> </name></person-group><article-title>Free-marginal multirater kappa (multirater &#x03BA;free): an alternative to Fleiss&#x2019; fixed-marginal multirater kappa</article-title><access-date>2026-08-23</access-date><conf-name>Joensuu Learning and Instruction Symposium 2005</conf-name><conf-date>Oct 14-15, 2005</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.academia.edu/2439155/Free_Marginal_Multirater_Kappa_multiraterfree_An_Alternative_to_Fleiss_Fixed_Marginal_Multirater_Kappa">https://www.academia.edu/2439155/Free_Marginal_Multirater_Kappa_multiraterfree_An_Alternative_to_Fleiss_Fixed_Marginal_Multirater_Kappa</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kolding</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lundin</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Hansen</surname><given-names>L</given-names> </name><name name-style="western"><surname>&#x00D8;stergaard</surname><given-names>SD</given-names> </name></person-group><article-title>Use of generative artificial intelligence (AI) in psychiatry and mental health care: a systematic review</article-title><source>Acta Neuropsychiatr</source><year>2024</year><month>11</month><day>11</day><volume>37</volume><fpage>e37</fpage><pub-id pub-id-type="doi">10.1017/neu.2024.50</pub-id><pub-id pub-id-type="medline">39523628</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Complete search strategy used in the Web of Science database.</p><media xlink:href="resprot_v15i1e91677_app1.docx" xlink:title="DOCX File, 10 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Data extraction form.</p><media xlink:href="resprot_v15i1e91677_app2.xlsx" xlink:title="XLSX File, 166 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>PRISMA-ScR checklist.</p><media xlink:href="resprot_v15i1e91677_app3.docx" xlink:title="DOCX File, 10 KB"/></supplementary-material></app-group></back></article>