<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">ResProt</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Res Protoc</journal-id>
      <journal-title>JMIR Research Protocols</journal-title>
      <issn pub-type="epub">1929-0748</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v15i1e93509</article-id>
      <article-id pub-id-type="pmid">42580683</article-id>
      <article-id pub-id-type="doi">10.2196/93509</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Protocol</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Protocol</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Methods of Evaluating Large Language Model–Based Health Care Applications Used by Nonprofessionals: Protocol for a Scoping Review</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Sarvestan</surname>
            <given-names>Javad</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Shah</surname>
            <given-names>Nigam</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Keuchel</surname>
            <given-names>Maren</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Health Care Informatics, Faculty of Health</institution>
            <institution>School of Medicine</institution>
            <institution>Witten/Herdecke University</institution>
            <addr-line>Pferdebachstrasse 11</addr-line>
            <addr-line>Witten, North Rhine-Westphalia, 58448</addr-line>
            <country>Germany</country>
            <phone>49 231 97677 333</phone>
            <email>maren.keuchel@uni-wh.de</email>
          </address>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0009-1611-6933</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Bisgin</surname>
            <given-names>Pinar</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5354-6766</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Strube</surname>
            <given-names>Tom</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-2216-894X</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Tschorn</surname>
            <given-names>Niklas</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0004-8706-9314</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Weltermann</surname>
            <given-names>Leoni</given-names>
          </name>
          <degrees>BSc</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0004-4187-0229</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Meister</surname>
            <given-names>Sven</given-names>
          </name>
          <degrees>Prof Dr</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-0522-986X</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Health Care Informatics, Faculty of Health</institution>
        <institution>School of Medicine</institution>
        <institution>Witten/Herdecke University</institution>
        <addr-line>Witten, North Rhine-Westphalia</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Department of Healthcare</institution>
        <institution>Fraunhofer Institute for Software and Systems Engineering</institution>
        <addr-line>Dortmund, North Rhine-Westphalia</addr-line>
        <country>Germany</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Maren Keuchel <email>maren.keuchel@uni-wh.de</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>11</day>
        <month>8</month>
        <year>2026</year>
      </pub-date>
      <volume>15</volume>
      <elocation-id>e93509</elocation-id>
      <history>
        <date date-type="received">
          <day>13</day>
          <month>2</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>29</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>31</day>
          <month>7</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>3</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Maren Keuchel, Pinar Bisgin, Tom Strube, Niklas Tschorn, Leoni Weltermann, Sven Meister. Originally published in JMIR Research Protocols (https://www.researchprotocols.org), 11.08.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Research Protocols, is properly cited. The complete bibliographic information, a link to the original publication on https://www.researchprotocols.org, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.researchprotocols.org/2026/1/e93509" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Large language models (LLMs) are increasingly used in health care by nonprofessionals (ie, individuals without formal training in health-related professions). These applications must be evaluated in an appropriate manner to prevent misinformation and harmful decisions. To date, guidance to evaluate LLM-based applications for nonprofessional users remains limited and fragmented, leaving researchers and developers without a scientifically grounded set of quality dimensions, metrics, and measurement tools to guide them.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This protocol outlines a scoping review that maps approaches for evaluation of LLM-based applications used for health purposes by nonprofessionals. It identifies current methods and maps them thematically by assigning them to evaluation dimensions, metrics, and measurement instruments. The review will provide a comprehensive overview of evaluation methods currently in use.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>The study follows the Joana Briggs Institute approach for conducting scoping reviews and reports. The protocol is reported in accordance with the PRISMA-P (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Protocols) guidelines, and the scoping review will be reported in accordance with the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) guidelines. The inclusion criteria comprise studies that evaluate LLM-based applications that are used in the context of health care by nonprofessionals. The search was conducted in PubMed, CINAHL, PsycInfo, and IEEE Xplore. Results since 2021 were considered. Data will be summarized and interpreted qualitatively. Publication screening was conducted by 2 independent reviewers in a blinded manner, with discrepancies settled through discussion. Data extraction and charting will be performed by 1 reviewer. To ensure quality, a random 10% sample of the publications will be independently charted by a second reviewer. Disagreements in the double-extracted subset will be resolved through discussion.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>As of July 2026, a steering committee of 6 researchers has been chosen for the conduct of the review. An initial search resulted in 8538 records after removing duplicates. After screening of these 8538 publications, 17.8% (1524/8538) were eligible for retrieval, of which 88.3% (1345/1524) were retrieved. Full-text screening (completed by 1 reviewer) excluded publications due to nonmatching populations (155/1345, 11.5%), concepts (246/1345, 18.3%), and contexts (24/1345, 1.8%), as well as secondary work (14/1345, 1%), leaving 67.4% (906/1345) of these publications for data extraction. We plan to perform final full-text screening, data extraction, coding, and synthesis of results in the fourth quarter of 2026.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>The scoping review aims to identify and map current evaluation methods for LLM-based applications used in health care by nonprofessionals. It will provide a systematic overview of the current state of research and insights into quality dimensions, metrics, and measurement instruments. The findings will provide directional guidance for further research and development in the field of quality assurance for LLM-based applications used by nonprofessionals.</p>
        </sec>
        <sec sec-type="registered-report">
          <title>International Registered Report Identifier (IRRID)</title>
          <p>DERR1-10.2196/93509</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>large language model</kwd>
        <kwd>LLM</kwd>
        <kwd>evaluation</kwd>
        <kwd>health care</kwd>
        <kwd>quality</kwd>
        <kwd>artificial intelligence</kwd>
        <kwd>AI</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <sec>
        <title>Background</title>
        <p>Large language model (LLM)–based health care applications are software systems that use LLMs to perform tasks that maintain and improve the health of populations and individuals [<xref ref-type="bibr" rid="ref1">1</xref>]. The number of LLM-based applications in the field of health care is rapidly increasing [<xref ref-type="bibr" rid="ref2">2</xref>]. They are being deployed across a diverse range of domains and use cases, such as chatbots answering medical questions, generation of patient information, clinical documentation, translation and summarization, and the creation of patient education material [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. Health care professionals such as physicians, nurses, or pharmacists use LLM-based applications mainly to support clinical decision-making [<xref ref-type="bibr" rid="ref5">5</xref>]. In this context, nonprofessionals are individuals that lack formal education with theoretical and factual knowledge on the diagnosis and treatment of health problems [<xref ref-type="bibr" rid="ref6">6</xref>]. They use LLM-based applications mainly to educate themselves on medical topics; interpret medical information; and receive lifestyle recommendations, support in customized medication use, perioperative care instructions, and support in physician-patient interaction [<xref ref-type="bibr" rid="ref7">7</xref>]. LLM-based applications offer promising solutions to many challenges faced by nonprofessionals. However, as this is still a relatively new field of research, there is also a high risk associated with their use in health care, especially when used by nonprofessionals as they may not recognize the potential for error or be able to check the plausibility of the recommendations made by LLM-based applications. For these reasons, careful evaluation of such systems is essential to ensure the safety and health of users.</p>
        <p>The European Union AI Act is a comprehensive legal framework for AI systems [<xref ref-type="bibr" rid="ref8">8</xref>]. It establishes a risk-based approach by assigning systems to risk classes, namely, unacceptable risk, high risk, limited risk, and minimal risk, and imposes corresponding requirements, with most obligations for providers of high-risk systems. Under the AI Act, systems are classified as high risk if their failure or malfunction could pose significant risks to an individual’s health, safety, or fundamental rights. Consequently, numerous applications within the medical domain are considered high risk and are therefore subject to stringent requirements. While the AI Act demands evaluation on transparency, human oversight, accuracy, robustness, and cybersecurity (chapter 3, articles 13-15), it does not specify how these should be measured or which standards should be applied.</p>
        <p>Previous studies have investigated evaluation methods of LLMs in clinical medicine and show that the metrics most examined are accuracy and consistency [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Benchmarks to evaluate the natural language understanding and generation of LLMs are Bilingual Evaluation Understudy [<xref ref-type="bibr" rid="ref11">11</xref>], Recall-Oriented Understudy for Gisting Evaluation [<xref ref-type="bibr" rid="ref12">12</xref>], Metric for Evaluation of Translation With Explicit Ordering, and BERTScore [<xref ref-type="bibr" rid="ref13">13</xref>]. To test LLMs’ ability to apply medical knowledge, the MedQA dataset [<xref ref-type="bibr" rid="ref14">14</xref>] is commonly used. It uses a questionnaire adapted from a standardized, multistage examination that physicians must pass to obtain a medical license in the United States [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. Other frequently used benchmarks to test medical knowledge are MultiMedQA [<xref ref-type="bibr" rid="ref17">17</xref>], PubMedQA [<xref ref-type="bibr" rid="ref18">18</xref>], and MedCaseReasoning [<xref ref-type="bibr" rid="ref19">19</xref>].</p>
        <p>Several publications point out the need for standardized evaluation frameworks to address clinical needs [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Others propose a systematic human approach for evaluating LLMs that support clinical tasks [<xref ref-type="bibr" rid="ref22">22</xref>] or frameworks to evaluate the clinical skills of LLM-based applications [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref26">26</xref>]. Most of this research focuses on the evaluation of clinical applications and use by health care professionals.</p>
      </sec>
      <sec>
        <title>Objectives</title>
        <p>This study aims to identify evaluation approaches of LLM-based applications used by nonprofessionals. The approaches are assigned to dimensions, sorted according to their metrics, and listed according to the measuring instruments used. The following research questions are addressed:</p>
        <list list-type="order">
          <list-item>
            <p>What are the quality dimensions of current evaluation approaches for LLM-based applications for nonprofessional users?</p>
          </list-item>
          <list-item>
            <p>How are these dimensions operationalized in terms of metrics and measurement instruments?</p>
          </list-item>
        </list>
      </sec>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Overview</title>
        <p>The Joana Briggs Institute methodology [<xref ref-type="bibr" rid="ref27">27</xref>] will be used as the overarching approach for conducting and documenting this scoping review. For the protocol, the PRISMA-P (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Protocols) [<xref ref-type="bibr" rid="ref28">28</xref>] guidelines were used. A completed PRISMA-P checklist for this protocol can be found in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The completed scoping review will be reported according to the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) [<xref ref-type="bibr" rid="ref29">29</xref>] guidelines.</p>
      </sec>
      <sec>
        <title>Protocol and Registration</title>
        <p>This protocol is authored by the research team and underwent peer review as a research protocol by JMIR Publications in July 2026 [<xref ref-type="bibr" rid="ref30">30</xref>]. This protocol was registered retrospectively with the Open Science Framework on July 24, 2026, after search and screening was substantially underway. Registration was initiated once the team determined that formal Open Science Framework registration would strengthen the transparency and reproducibility of the review. No deviations from the registered protocol occurred during the search and screening phase of the review. The codebook for classifying evaluation dimensions remains subject to iterative refinement during data extraction and synthesis. Any substantive changes will be reported in the final manuscript.</p>
      </sec>
      <sec>
        <title>Eligibility Criteria</title>
        <p>The population, concept, and context scheme [<xref ref-type="bibr" rid="ref31">31</xref>] was used to define the inclusion criteria. Additional criteria concerning the classification of source types were introduced. Exclusion criteria were defined to complement the inclusion criteria and enhance clarity in the screening process. <xref ref-type="table" rid="table1">Table 1</xref> shows the inclusion and exclusion criteria and their mapping to the population, concept, and context framework. The restricted inclusion to publications from the last 5 years (from January 2021 to January 2026) was chosen due to the fast-paced innovations in this field of research. This time frame includes the launch of ChatGPT and other important publicly available chatbots. The language restrictions to English and German were chosen due to resource constraints in personnel and language competence of the authors.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Inclusion and exclusion criteria and their reference to the population, concept, and context schema.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="160"/>
            <col width="480"/>
            <col width="360"/>
            <thead>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Inclusion criteria</td>
                <td>Exclusion criteria</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>General</td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Published in 2021-2025</p>
                    </list-item>
                    <list-item>
                      <p>English or German language</p>
                    </list-item>
                    <list-item>
                      <p>Primary studies with own data collection</p>
                    </list-item>
                    <list-item>
                      <p>Full text available</p>
                    </list-item>
                  </list>
                </td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Published before 2021</p>
                    </list-item>
                    <list-item>
                      <p>Other languages</p>
                    </list-item>
                    <list-item>
                      <p>Secondary work without own data collection; comments</p>
                    </list-item>
                    <list-item>
                      <p>No full-text access</p>
                    </list-item>
                  </list>
                </td>
              </tr>
              <tr valign="top">
                <td>Population</td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Explicitly used by or intended to be used by individuals without formal training in health care professions (eg, patients and relatives), including systems designed for dual use by professional health care providers and laypersons; this includes direct prompting from the patient perspective (“I have pain in...”) and indirect questioning (“Supply important patient information for...”)</p>
                    </list-item>
                  </list>
                </td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Exclusive use by medical personnel or other professional health care providers or no specific description of the target audience</p>
                    </list-item>
                  </list>
                </td>
              </tr>
              <tr valign="top">
                <td>Concept</td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Technology: LLM<sup>a</sup>-based systems, transformer language models, and RAG<sup>b</sup>; this comprises both proprietary systems that interface with LLMs via APIs and publicly accessible LLM chatbots, including specialized medical chatbots and general-purpose chatbots that are queried for medical purposes</p>
                    </list-item>
                    <list-item>
                      <p>Evaluation: evaluation mechanisms described</p>
                    </list-item>
                  </list>
                </td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Exclusively rule-based chatbots or other systems without the use of LLMs</p>
                    </list-item>
                    <list-item>
                      <p>No evaluation mechanisms; purely architectural descriptions</p>
                    </list-item>
                  </list>
                </td>
              </tr>
              <tr valign="top">
                <td>Context</td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Health care, prevention, health navigation, patient education, shared decision-making, self-diagnosis, symptom checking, and self-triage</p>
                    </list-item>
                  </list>
                </td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>No reference to health care or an exclusively professional context, clinical workflows, clinical research, triage, and clinical decision-making without active involvement of patients</p>
                    </list-item>
                  </list>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>LLM: large language model.</p>
            </fn>
            <fn id="table1fn2">
              <p><sup>b</sup>RAG: retrieval-augmented generation.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Information Sources</title>
        <p>The following databases were selected to comprehensively cover the interdisciplinary scope of the review. Medical and health science databases (PubMed, CINAHL, and PsycInfo) were consulted to capture literature on digital health, patient engagement, and measurement instruments. The IEEE Xplore database specializes in technology and computer science and was included to ensure coverage of research on LLMs, system architectures, and evaluation methodologies. The use of generative AI for literature research was transparently documented (model, version, prompts, and date) and made reproducible.</p>
        <p>Publications in both German and English were considered. To find further relevant sources, a backward citation search for included conference papers was performed after completion of screening, in which all references of the included sources were screened using the same inclusion criteria. By reviewing the sources of the included papers as part of the full-text screening process, conferences referenced in those sources were also included.</p>
      </sec>
      <sec>
        <title>Search Strategy</title>
        <p>In the first step, an initial search was conducted on December 1, 2025, in PubMed and IEEE Xplore using the search string “Large Language Model Evaluation health care,” yielding 1300 and 119 results, respectively. The first 50 results from each database were screened. Titles, abstracts, and index terms of these potentially relevant publications were analyzed to identify additional search terms. The resulting concepts and synonyms are summarized in <xref ref-type="table" rid="table2">Table 2</xref>.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Concepts and their related terms.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="180"/>
            <col width="820"/>
            <thead>
              <tr valign="top">
                <td>Concept</td>
                <td>Related terms</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Evaluation</td>
                <td>“Assessment,” “validation,” “appraisal,” “effectiveness,” “usability,” “performance,” “safety,” “acceptability,” “user experience,” “evidence,” “objective metrics,” and “review methods”</td>
              </tr>
              <tr valign="top">
                <td>Large language model</td>
                <td>“LLM,” “chatbot,” “GPT,” “ChatGPT,” “generative AI,” “genAI,” and “foundation models”</td>
              </tr>
              <tr valign="top">
                <td>Health care</td>
                <td>“Health care,” “clinical care,” “medical care,” “digital health,” “patient care,” “patient information,” “patient education,” “patient management,” “patient assistance,” “personal health,” and “medical”</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>The search terms were combined to construct the search strings using the “OR” operator within individual concepts and the “AND” operator between different concepts. The strings were designed to search for the specified terms in titles and abstracts and adapted as required for each database. <xref ref-type="boxed-text" rid="box1">Textbox 1</xref> provides an example of the PubMed search string, where “[tiab]” indicates a title and abstract search. For this search string, MeSH terms [<xref ref-type="bibr" rid="ref32">32</xref>] were used to capture established concepts, and free-text terms were used to include emerging topics and topics that were not covered by the MeSH library. The complete search strings for all databases are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p>
        <boxed-text id="box1" position="float">
          <title>PubMed search string.</title>
          <p>( “Evaluation Studies as Topic”[Mesh] OR Assessment[tiab] OR Validation[tiab] OR Appraisal[tiab] OR Effectiveness[tiab] OR Usability[tiab] OR Performance[tiab] OR “Safety” [Mesh] OR “Patient Acceptance of Health Care” [Mesh] OR “User Experience”[tiab] OR Evidence[tiab] OR “Objective Metrics”[tiab] OR “Review Methods”[tiab]) AND (“Large Language Models”[mesh] OR LLM[tiab] OR Chatbot*[tiab] OR GPT[tiab] OR ChatGPT[tiab] OR “Generative Artificial Intelligence”[Mesh] OR “Foundation Models”[tiab]) AND ( “Delivery of Health Care”[Mesh] OR “Clinical Care”[tiab] OR “Medical Care”[tiab] OR “Digital Health”[Mesh] OR “Patient Care”[Mesh] OR “Patient Information” [tiab] OR “Patient Education as Topic” [Mesh] OR “Patient Care Management”[Mesh] OR “Patient Assistance”[tiab] OR “Personal Health Services”[Mesh] OR “Medical”)</p>
        </boxed-text>
      </sec>
      <sec>
        <title>Screening</title>
        <p>To achieve high interrater reliability, a set of 50 publications was screened by all reviewers, and the results were discussed to resolve discrepancies. Studies were selected based on the above-mentioned inclusion criteria and were reviewed in 2 stages. In the first stage, titles and abstracts were reviewed, followed by the full texts. Six reviewers screened the titles, abstracts, and full texts of the publications in pairs. This process was conducted in a blinded manner, meaning that reviewers could not see the evaluations provided by others while screening the publications. In the event of disagreements, a third reviewer was consulted. Remaining disagreements regarding study selection and data extraction were resolved through discussion and consensus with the review team. The web-based tool Rayyan (Rayyan Systems Inc) [<xref ref-type="bibr" rid="ref33">33</xref>] was used to screen publications and manage resources. Rayyan provides AI-assisted support, which in this study was used to identify and remove duplicates and highlight key terms within publications. Additionally, Rayyan makes suggestions for inclusion or exclusion. The software interface with the aforementioned functions is shown in <xref rid="figure1" ref-type="fig">Figure 1</xref>.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Screenshot of screening software Rayyan showing the highlighting of keywords and suggestions for decisions.</p>
          </caption>
          <graphic xlink:href="resprot_v15i1e93509_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Data Charting</title>
        <p>Included publications will be analyzed, and data will be extracted in tabular form. A template with PRISMA-ScR–compliant data items adapted from a methodological guide for scoping reviews [<xref ref-type="bibr" rid="ref31">31</xref>] will be used as an overall starting framework. The selection of data fields is preliminary and will be iteratively refined by the review team if additional relevant information dimensions are identified in the included publications. Data extraction and charting will be performed by 1 reviewer. To ensure quality, a random sample of 20 publications will be independently charted by a second reviewer for calibration. Interrater reliability will be assessed using the Cohen κ, with a prespecified threshold above 0.61 required among reviewers in this subsample. If this threshold is not met, reviewers will undergo a recalibration process where discrepant ratings will be discussed and category definitions will be refined. A new sample of 20 publications will be independently rerated, and the charting process and data items will be iteratively refined until the threshold is achieved.</p>
      </sec>
      <sec>
        <title>Data Items</title>
        <p>The preliminary data items are (1) reference (title, author, journal, year, and page), (2) study type, (3) population, (4) context (health care task), (5) type of technology used (LLM name and version), (6) evaluation dimension, (7) metric, (8) measuring instrument, and (9) use of reporting frameworks (such as CONSORT-AI [Consolidated Standards of Reporting Trials–AI] [<xref ref-type="bibr" rid="ref34">34</xref>]).</p>
      </sec>
      <sec>
        <title>Coding</title>
        <p>The data charting step will be followed by the coding of quality dimensions, metrics, and measurement instruments using a deductive-inductive coding scheme with iterative refinement of the coding book based on the principles of qualitative content analysis by Mayring [<xref ref-type="bibr" rid="ref35">35</xref>]. While the evaluation items will be previously extracted verbatim from the publications, they will be subsequently coded to superordinate dimensions. The coding and calibration process will follow a stepwise approach with iterative refinement of the codebook. The initial codebook for evaluation dimensions uses the categories suggested by the AI Risk Management Framework (RMF) [<xref ref-type="bibr" rid="ref36">36</xref>] for AI risks and trustworthiness. The AI RMF suggests seven dimensions of risks and trustworthiness: (1) validity and reliability, (2) safety, (3) security and resilience, (4) accountability and transparency, (5) explainability and interpretability, (6) privacy enhancement, and (7) fairness.</p>
        <p>In the first step of pilot coding, these dimensions will be used as categories and charted as present or not present for each publication (deductive step). To chart the metrics used for each quality dimension, the exact wording in the publication will be used (eg, “accuracy” for the first dimension). The measuring instrument will describe the actual evaluation method used (eg, “5-point ordinal scale”) and the evaluating instance (eg, “medical professional” or “software”). An example is provided in <xref ref-type="table" rid="table3">Table 3</xref>. Metrics reported in studies that cannot be mapped to the predefined categories provided by the AI RMF will be charted as possible emergent categories. The pilot coding will include a random set of 20 publications, which will be double screened by 2 independent coders. Following the pilot phase, coders will compare results and discuss disagreements and emergent dimensions. Newly identified categories will be added to the codebook if they represent a distinct concept not encompassed by any existing dimension, and both raters will agree on whether the emergent category is sufficiently supported by data. Subsequently, to ensure reliability, the codebook will undergo iterative calibration rounds. In each calibration round, both coders will independently code a new random set of 20 publications. Intercoder reliability will be measured using the Cohen κ. Disagreements will be resolved through discussion, leading to further refinement of the codebook. The calibration process will continue until the Cohen κ reaches 0.61, reflecting substantial agreement between coders. Once calibration achieves the reliability threshold, the finalized codebook will be used for the primary coding of the remaining publications. To enhance efficiency, coding will be performed by a single coder following the established guidelines in the codebook. A random subset of 10% will undergo additional independent coding by the second rater to confirm reliability during main coding. The coding process will implement transparency via systematic documentation of all adjustments to dimensions, inclusions of new categories, and discussions.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Example of data charting of dimensions, metrics, and measuring instruments.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="320"/>
            <col width="130"/>
            <col width="170"/>
            <col width="380"/>
            <thead>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Dimension</td>
                <td>Metric</td>
                <td>Measuring instrument</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Validity and reliability</td>
                <td>Yes</td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Accuracy</p>
                    </list-item>
                    <list-item>
                      <p>Reliability</p>
                    </list-item>
                  </list>
                </td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Accuracy: 5-point ordinal scale</p>
                    </list-item>
                    <list-item>
                      <p>Reliability: 3-point ordinal scale</p>
                    </list-item>
                    <list-item>
                      <p>Both rated by 2 medical professionals</p>
                    </list-item>
                  </list>
                </td>
              </tr>
              <tr valign="top">
                <td>Safety</td>
                <td>No</td>
                <td>—<sup>a</sup></td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>Explainability and interpretability</td>
                <td>Yes</td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Readability</p>
                    </list-item>
                  </list>
                </td>
                <td>
                  <list list-type="bullet">
                    <list-item>
                      <p>Flesch-Kincaid reading grade level</p>
                    </list-item>
                    <list-item>
                      <p>Automated via web tool</p>
                    </list-item>
                  </list>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>Not applicable</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Critical Appraisal</title>
        <p>There is no formal quality assessment planned (Joana Briggs Institute compliant) as the goal is to map the evidence field.</p>
      </sec>
      <sec>
        <title>Synthesis of Results</title>
        <p>A PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flowchart will be generated to illustrate the various reasons for excluding publications. A template from PRISMA will be used for this purpose and will be available in the publication of the scoping review results. Charted data will be managed according to the a priori–defined data items. Data will be aggregated across studies and summarized using descriptive statistics and structured narrative synthesis. A deductive-inductive coding scheme will be used to determine the quality dimensions.</p>
        <p>The results will be presented in tables, figures, and a summary description to illustrate the scope and characteristics of the evaluation mechanisms used. The review will be reported in accordance with PRISMA-ScR guidelines.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <p>In January 2026, a steering committee of 6 researchers was established to carry out the review. A search performed on January 20, 2026, on all databases yielded 9311 results, including 60.9% (5671/9311) results from PubMed, 26.4% (2459/9311) results from IEEE Xplore, 9.5% (885/9311) results from CINAHL, and 3.2% (296/9311) results from PsycInfo. Of these 9311 publications, after removing 8.3% (773/9311) duplicates, 91.7% (8538/9311) remained for screening. As of July 2026, after screening, 1524 publications remained for retrieval, from which 88.3% (1345/1524) records were retrieved. In the following step of full-text screening, of the 1345 retrieved records, 11.5% (155/1345) were excluded due to nonmatching populations (see details in the Eligibility Criteria section), 18.3% (246/1345) were excluded due to nonmatching concepts (other technology or no evaluation described), 1.8% (24/1345) were excluded due to nonmatching contexts, and 1% (14/1345) were excluded due to being secondary work. This resulted in 906 publications remaining for data charting. This count is provisional pending second-reviewer verification in accordance with the protocol. Data collection, coding and synthesis of the results had not been started as of August 5, 2026. The expected start for these 2 phases is the middle of August 2026. The PRISMA flowchart in <xref rid="figure2" ref-type="fig">Figure 2</xref> [<xref ref-type="bibr" rid="ref37">37</xref>] shows the number of studies that were identified, screened, and included.</p>
      <fig id="figure2" position="float">
        <label>Figure 2</label>
        <caption>
          <p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flowchart of records in the phases of identification, screening, and inclusion (template from Page et al [<xref ref-type="bibr" rid="ref37">37</xref>]).</p>
        </caption>
        <graphic xlink:href="resprot_v15i1e93509_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
      </fig>
      <p>We expect the final full-text screening, data extraction, coding, and synthesis results for the fourth quarter of 2026.</p>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Anticipated Findings</title>
        <p>The expected outcome of this study is primarily an overview of mechanisms for evaluating LLM systems used in health care by nonprofessionals. In this process, certain quality dimensions, most notably accuracy, are likely to emerge as prominent, having been examined across numerous studies, whereas other dimensions are investigated in only a limited subset of the literature. We expect to identify a set of dimensions that are evaluated in current publications and the degree of alignment with current definition categories. The outcome includes metrics used to describe the specific matter evaluated in each dimension. We also expect a comprehensive overview of the measuring instruments used to evaluate each metric. We anticipate finding dominant or underrepresented dimensions and metrics and identify common operationalizations for measurements.</p>
      </sec>
      <sec>
        <title>Comparison With Prior Work</title>
        <p>This study builds on previous work and extends existing approaches to categorizing and mapping evaluation mechanisms by incorporating the specific perspective of use by nonprofessionals. Prior work has largely emphasized systems facing health care professionals, whereas our focus allows for insights into how quality dimensions are operationalized by nonprofessional users for health care queries.</p>
      </sec>
      <sec>
        <title>Dissemination Plan</title>
        <p>The findings of this review will be disseminated through multiple channels to different stakeholder groups. Results will be submitted to a peer-reviewed journal and as international conference papers to connect with the scientific community. A preprint of this research protocol and the subsequent research paper will be made available online. To reach practitioners and experts, open workshops will be held to discuss the practical applicability and relevance of an evaluation guideline for LLM chatbots in health care. The research findings will be adapted for the general public in summaries that are accessible to nonexperts and published by the associated organizations of the authors on social media platforms.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>A general limitation of scoping reviews is their lack of strict quality assessment of sources. Another limitation is the exclusion of non–English- or non–German-language publications; this may overlook contributions from research groups working in other languages. Given the prevalence of publication bias in scientific research, the predominance of positive results in this study may likewise skew the assessment of the actual situation. The absence of a quality assessment limits the derivation of efficacy judgments. The results will serve to map the evidence and generate hypotheses.</p>
        <p>Furthermore, the rapid evolution of LLMs implies that the findings may become outdated by the time of publication, particularly when the publication process involves lengthy peer review.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>The scoping review will provide a comprehensive overview of evaluation methods used in LLM applications for nonprofessionals in health care. It will show the range of evaluation methods and explain quality dimensions with their operationalization mechanisms applied in scientific work. It will identify research gaps and provide directional guidance for further research and development in the field of quality assurance of LLM applications for nonprofessional users.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Complete list of final search strings for all databases.</p>
        <media xlink:href="resprot_v15i1e93509_app1.pdf" xlink:title="PDF File  (Adobe PDF File), 146 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>PRISMA-P checklist.</p>
        <media xlink:href="resprot_v15i1e93509_app2.pdf" xlink:title="PDF File  (Adobe PDF File), 223 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">CONSORT-AI</term>
          <def>
            <p>Consolidated Standards of Reporting Trials–Artificial Intelligence</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">PRISMA</term>
          <def>
            <p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">PRISMA-P</term>
          <def>
            <p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses Protocols</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">PRISMA-ScR</term>
          <def>
            <p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">RAG</term>
          <def>
            <p>retrieval-augmented generation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">RMF</term>
          <def>
            <p>Risk Management Framework</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors declare the use of generative AI (GenAI) in the research and writing process. According to the Generative AI Delegation Taxonomy (2025), the following tasks were delegated to GenAI tools under full human supervision: data curation and organization, translation, development of experimental or research protocols, and proofreading and editing. For data curation and organization, the GenAI tool used was the Rayyan proprietary abstract screening tool (Rayyan Systems Inc). The authors used this tool for supplementary assessment of abstracts for fit with the eligibility criteria. For translation, the GenAI tool used was the DeepL translation tool [<xref ref-type="bibr" rid="ref38">38</xref>]. Sections of the paper were drafted in the authors’ native language (German), translated using DeepL, reviewed by the authors, and then transferred to the manuscript. For development of experimental or research protocols and proofreading and editing, the GenAI tool used was Claude Opus 4.8 R (Anthropic). The authors used Claude to assist with the development of the codebook for mapping quality dimensions. It was used to search for appropriate reference documents and screening for grammatical errors and logical inconsistencies in the protocol manuscript. Responsibility for the final manuscript lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes. No other GenAI tools were used at any stage of manuscript development or screening of data.</p>
    </ack>
    <notes>
      <title>Funding</title>
      <p>No financial support or grants were received from any public, commercial, or not-for-profit entities for the research, authorship, or publication of this article.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>MK contributed to conceptualization, writing (original draft), and writing (review and editing). PB, TS, NT, and LW contributed to writing (review and editing). SM contributed to supervision.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Maity</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Saikia</surname>
              <given-names>MJ</given-names>
            </name>
          </person-group>
          <article-title>Large language models in healthcare and medical applications: a review</article-title>
          <source>Bioengineering (Basel)</source>
          <year>2025</year>
          <month>06</month>
          <day>10</day>
          <volume>12</volume>
          <issue>6</issue>
          <fpage>631</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=bioengineering12060631"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/bioengineering12060631</pub-id>
          <pub-id pub-id-type="medline">40564447</pub-id>
          <pub-id pub-id-type="pii">bioengineering12060631</pub-id>
          <pub-id pub-id-type="pmcid">PMC12189880</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Carchiolo</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Malgeri</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Trends, challenges, and applications of large language models in healthcare: a bibliometric and scoping review</article-title>
          <source>Future Internet</source>
          <year>2025</year>
          <month>02</month>
          <day>08</day>
          <volume>17</volume>
          <issue>2</issue>
          <fpage>76</fpage>
          <pub-id pub-id-type="doi">10.3390/fi17020076</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Busch</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hoffmann</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Rueger</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>van Dijk</surname>
              <given-names>EH</given-names>
            </name>
            <name name-style="western">
              <surname>Kader</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ortiz-Prado</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Makowski</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Saba</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Hadamitzky</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kather</surname>
              <given-names>JN</given-names>
            </name>
            <name name-style="western">
              <surname>Truhn</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Cuocolo</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Adams</surname>
              <given-names>LC</given-names>
            </name>
            <name name-style="western">
              <surname>Bressem</surname>
              <given-names>KK</given-names>
            </name>
          </person-group>
          <article-title>Current applications and challenges in large language models for patient care: a systematic review</article-title>
          <source>Commun Med (Lond)</source>
          <year>2025</year>
          <month>01</month>
          <day>21</day>
          <volume>5</volume>
          <issue>1</issue>
          <fpage>26</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s43856-024-00717-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s43856-024-00717-2</pub-id>
          <pub-id pub-id-type="medline">39838160</pub-id>
          <pub-id pub-id-type="pii">10.1038/s43856-024-00717-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC11751060</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rust</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Frings</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Meister</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Fehring</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of a large language model to simplify discharge summaries and provide cardiological lifestyle recommendations</article-title>
          <source>Commun Med (Lond)</source>
          <year>2025</year>
          <month>05</month>
          <day>29</day>
          <volume>5</volume>
          <issue>1</issue>
          <fpage>208</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s43856-025-00927-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s43856-025-00927-2</pub-id>
          <pub-id pub-id-type="medline">40442348</pub-id>
          <pub-id pub-id-type="pii">10.1038/s43856-025-00927-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC12122782</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Nezhad</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Hosseini</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Zolnour</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Zonour</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Hosseini</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Topaz</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zolnoori</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>A scoping review of large language model applications in healthcare</article-title>
          <source>Stud Health Technol Inform</source>
          <year>2025</year>
          <month>08</month>
          <day>07</day>
          <volume>329</volume>
          <fpage>1966</fpage>
          <lpage>7</lpage>
          <pub-id pub-id-type="doi">10.3233/SHTI251302</pub-id>
          <pub-id pub-id-type="medline">40776319</pub-id>
          <pub-id pub-id-type="pii">SHTI251302</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="web">
          <article-title>Classifying health workers: mapping occupations to the International Standard Classification</article-title>
          <source>World Health Organization</source>
          <year>2019</year>
          <month>07</month>
          <day>31</day>
          <access-date>2026-07-31</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.who.int/publications/m/item/classifying-health-workers">https://www.who.int/publications/m/item/classifying-health-workers</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Aydin</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Karabacak</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Vlachos</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Margetis</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Large language models in patient education: a scoping review of applications in medicine</article-title>
          <source>Front Med (Lausanne)</source>
          <year>2024</year>
          <month>10</month>
          <day>29</day>
          <volume>11</volume>
          <fpage>1477898</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/fmed.2024.1477898"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fmed.2024.1477898</pub-id>
          <pub-id pub-id-type="medline">39534227</pub-id>
          <pub-id pub-id-type="pmcid">PMC11554522</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="web">
          <article-title>Regulation (EU) 2024/1689 of the European Parliament and of the Council of 13 June 2024 laying down harmonised rules on artificial intelligence and amending Regulations (EC) No 300/2008, (EU) No 167/2013, (EU) No 168/2013, (EU) 2018/858, (EU) 2018/1139 and (EU) 2019/2144 and Directives 2014/90/EU, (EU) 2016/797 and (EU) 2020/1828 (Artificial Intelligence Act) (Text with EEA relevance)</article-title>
          <source>European Union</source>
          <year>2024</year>
          <access-date>2026-01-07</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://eur-lex.europa.eu/eli/reg/2024/1689/oj/eng">https://eur-lex.europa.eu/eli/reg/2024/1689/oj/eng</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shool</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Adimi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Saboori Amleshi</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Bitaraf</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Golpira</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Tara</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>A systematic review of large language model (LLM) evaluations in clinical medicine</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2025</year>
          <month>03</month>
          <day>07</day>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>117</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-025-02954-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-025-02954-4</pub-id>
          <pub-id pub-id-type="medline">40055694</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-025-02954-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC11889796</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bedi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Orr-Ewing</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Dash</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Koyejo</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Callahan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Fries</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Wornow</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Swaminathan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lehmann</surname>
              <given-names>LS</given-names>
            </name>
            <name name-style="western">
              <surname>Hong</surname>
              <given-names>HJ</given-names>
            </name>
            <name name-style="western">
              <surname>Kashyap</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Chaurasia</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NR</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tazbaz</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Milstein</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pfeffer</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NH</given-names>
            </name>
          </person-group>
          <article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title>
          <source>JAMA</source>
          <year>2025</year>
          <month>01</month>
          <day>28</day>
          <volume>333</volume>
          <issue>4</issue>
          <fpage>319</fpage>
          <lpage>28</lpage>
          <pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id>
          <pub-id pub-id-type="medline">39405325</pub-id>
          <pub-id pub-id-type="pii">2825147</pub-id>
          <pub-id pub-id-type="pmcid">PMC11480901</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Papineni</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Roukos</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ward</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>WJ</given-names>
            </name>
          </person-group>
          <article-title>BLEU: a method for automatic evaluation of machine translation</article-title>
          <source>Proceedings of the 40th Annual Meeting on Association for Computational Linguistics</source>
          <year>2002</year>
          <conf-name>ACL '02</conf-name>
          <conf-date>Jul 7-12, 2002</conf-date>
          <conf-loc>Philadelphia, PA</conf-loc>
          <pub-id pub-id-type="doi">10.3115/1073083.1073135</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>CY</given-names>
            </name>
          </person-group>
          <article-title>ROUGE: a package for automatic evaluation of summaries</article-title>
          <source>Text Summarization Branches Out</source>
          <year>2004</year>
          <publisher-loc>Stroudsburg, PA</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Kishore</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Weinberger</surname>
              <given-names>KQ</given-names>
            </name>
            <name name-style="western">
              <surname>Artzi</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>BERTScore: evaluating text generation with BERT</article-title>
          <source>arXiv. Preprint posted online on April 21, 2019</source>
          <year>2026</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.1904.09675</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Pan</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Oufattole</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Weng</surname>
              <given-names>WH</given-names>
            </name>
            <name name-style="western">
              <surname>Fang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Szolovits</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>What disease does this patient have? A large-scale open domain question answering dataset from medical exams</article-title>
          <source>Appl Sci</source>
          <year>2021</year>
          <month>07</month>
          <day>12</day>
          <volume>11</volume>
          <issue>14</issue>
          <fpage>6421</fpage>
          <pub-id pub-id-type="doi">10.3390/app11146421</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Siam</surname>
              <given-names>MK</given-names>
            </name>
            <name name-style="western">
              <surname>Varela</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Faruk</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>JQ</given-names>
            </name>
            <name name-style="western">
              <surname>Gu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Maruf</surname>
              <given-names>AA</given-names>
            </name>
            <name name-style="western">
              <surname>Aung</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>Benchmarking large language models on the United States Medical Licensing Examination for clinical reasoning and medical licensing scenarios</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <month>12</month>
          <day>03</day>
          <volume>16</volume>
          <issue>1</issue>
          <fpage>1387</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-025-31010-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-025-31010-4</pub-id>
          <pub-id pub-id-type="medline">41339739</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-025-31010-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC12796295</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="web">
          <article-title>About the USMLE</article-title>
          <source>United States Medical Licensing Examination</source>
          <access-date>2026-01-12</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.usmle.org/about-usmle">https://www.usmle.org/about-usmle</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tanwani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Seneviratne</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gamble</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Babiker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schärli</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhery</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tomasev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>08</month>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>80</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Dhingra</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Cohen</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>PubMedQA: a dataset for biomedical research question answering</article-title>
          <source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing</source>
          <year>2019</year>
          <conf-name>EMNLP-IJCNLP 2019</conf-name>
          <conf-date>Nov 3-7, 2019</conf-date>
          <conf-loc>Hong Kong, China</conf-loc>
          <pub-id pub-id-type="doi">10.18653/v1/D19-1259</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Thapa</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Suresh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Tao</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lozano</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>MedCaseReasoning: evaluating and learning diagnostic reasoning from clinical case reports</article-title>
          <source>arXiv. Preprint posted online on May 16, 2025</source>
          <year>2026</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.2505.11733</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Park</surname>
              <given-names>YJ</given-names>
            </name>
            <name name-style="western">
              <surname>Pillai</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Deng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Gupta</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Paget</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Naugler</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Assessing the research landscape and clinical utility of large language models: a scoping review</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2024</year>
          <month>03</month>
          <day>12</day>
          <volume>24</volume>
          <issue>1</issue>
          <fpage>72</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-024-02459-6"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-024-02459-6</pub-id>
          <pub-id pub-id-type="medline">38475802</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-024-02459-6</pub-id>
          <pub-id pub-id-type="pmcid">PMC10936025</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Shin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Cho</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Analyzing evaluation methods for large language models in the medical field: a scoping review</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2024</year>
          <month>11</month>
          <day>29</day>
          <volume>24</volume>
          <issue>1</issue>
          <fpage>366</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-024-02709-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-024-02709-7</pub-id>
          <pub-id pub-id-type="medline">39614219</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-024-02709-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11606129</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tam</surname>
              <given-names>TY</given-names>
            </name>
            <name name-style="western">
              <surname>Sivarajkumar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kapoor</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Stolyar</surname>
              <given-names>AV</given-names>
            </name>
            <name name-style="western">
              <surname>Polanska</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>McCarthy</surname>
              <given-names>KR</given-names>
            </name>
            <name name-style="western">
              <surname>Osterhoudt</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Visweswaran</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Fu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mathur</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Cacciamani</surname>
              <given-names>GE</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title>
          <source>NPJ Digit Med</source>
          <year>2024</year>
          <month>09</month>
          <day>28</day>
          <volume>7</volume>
          <issue>1</issue>
          <fpage>258</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-024-01258-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id>
          <pub-id pub-id-type="medline">39333376</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-024-01258-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11437138</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gupta</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Aizawa</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>RR</given-names>
            </name>
          </person-group>
          <article-title>Med-CoDE: medical critique based disagreement evaluation framework</article-title>
          <source>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies</source>
          <year>2025</year>
          <publisher-loc>Stroudsburg, PA</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Bian</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Jang</surname>
              <given-names>WS</given-names>
            </name>
            <name name-style="western">
              <surname>Ouyang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>MedQA-CS: benchmarking large language models clinical skills using an AI-SCE framework</article-title>
          <source>arXiv. Preprint posted online on October 2, 2024</source>
          <year>2026</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/html/2410.01553v1"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Fragiadakis</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Diou</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Kousiouris</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Nikolaidou</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Evaluating human-AI collaboration: a review and methodological framework</article-title>
          <source>arXiv. Preprint posted online on July 9, 2024</source>
          <year>2026</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.2407.19098</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kanithi</surname>
              <given-names>PK</given-names>
            </name>
            <name name-style="western">
              <surname>Christophe</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Pimentel</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Raha</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Saadi</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Javed</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Maslenkova</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hayat</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Rajan</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Khan</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>MEDIC: towards a comprehensive framework for evaluating LLMs in clinical applications</article-title>
          <source>arXiv. Preprint posted online on September 11, 2024</source>
          <year>2026</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2409.07314v1"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2409.07314</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Peters</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Marnie</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Tricco</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Pollock</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Munn</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Alexander</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>McInerney</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Godfrey</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Khalil</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Updated methodological guidance for the conduct of scoping reviews</article-title>
          <source>JBI Evid Synth</source>
          <year>2020</year>
          <month>10</month>
          <volume>18</volume>
          <issue>10</issue>
          <fpage>2119</fpage>
          <lpage>26</lpage>
          <pub-id pub-id-type="doi">10.11124/JBIES-20-00167</pub-id>
          <pub-id pub-id-type="medline">33038124</pub-id>
          <pub-id pub-id-type="pii">02174543-202010000-00004</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Moher</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Shamseer</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Clarke</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ghersi</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Liberati</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Petticrew</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Shekelle</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Stewart</surname>
              <given-names>LA</given-names>
            </name>
          </person-group>
          <article-title>Preferred Reporting Items for Systematic review and Meta-Analysis Protocols (PRISMA-P) 2015 statement</article-title>
          <source>Syst Rev</source>
          <year>2015</year>
          <month>01</month>
          <day>01</day>
          <volume>4</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://systematicreviewsjournal.biomedcentral.com/articles/10.1186/2046-4053-4-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/2046-4053-4-1</pub-id>
          <pub-id pub-id-type="medline">25554246</pub-id>
          <pub-id pub-id-type="pii">2046-4053-4-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC4320440</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tricco</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Lillie</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Zarin</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>O'Brien</surname>
              <given-names>KK</given-names>
            </name>
            <name name-style="western">
              <surname>Colquhoun</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Levac</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Moher</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Peters</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Horsley</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Weeks</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Hempel</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Akl</surname>
              <given-names>EA</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>McGowan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Stewart</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Hartling</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Aldcroft</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wilson</surname>
              <given-names>MG</given-names>
            </name>
            <name name-style="western">
              <surname>Garritty</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lewin</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Godfrey</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Macdonald</surname>
              <given-names>MT</given-names>
            </name>
            <name name-style="western">
              <surname>Langlois</surname>
              <given-names>EV</given-names>
            </name>
            <name name-style="western">
              <surname>Soares-Weiser</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Moriarty</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Clifford</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Tunçalp</surname>
              <given-names>Ö</given-names>
            </name>
            <name name-style="western">
              <surname>Straus</surname>
              <given-names>SE</given-names>
            </name>
          </person-group>
          <article-title>PRISMA extension for Scoping Reviews (PRISMA-ScR): checklist and explanation</article-title>
          <source>Ann Intern Med</source>
          <year>2018</year>
          <month>10</month>
          <day>02</day>
          <volume>169</volume>
          <issue>7</issue>
          <fpage>467</fpage>
          <lpage>73</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.acpjournals.org/doi/10.7326/M18-0850?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.7326/M18-0850</pub-id>
          <pub-id pub-id-type="medline">30178033</pub-id>
          <pub-id pub-id-type="pii">2700389</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="web">
          <article-title>JMIR research protocols</article-title>
          <source>JMIR Publications</source>
          <access-date>2025-12-18</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.researchprotocols.org/">https://www.researchprotocols.org/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>von Elm</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Schreiber</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Haupt</surname>
              <given-names>CC</given-names>
            </name>
          </person-group>
          <article-title>Methodische anleitung für scoping reviews (JBI-Methodologie) [Article in German]</article-title>
          <source>Z Evid Fortbild Qual Gesundhwes</source>
          <year>2019</year>
          <month>06</month>
          <volume>143</volume>
          <fpage>1</fpage>
          <lpage>7</lpage>
          <pub-id pub-id-type="doi">10.1016/j.zefq.2019.05.004</pub-id>
          <pub-id pub-id-type="medline">31296451</pub-id>
          <pub-id pub-id-type="pii">S1865-9217(19)30066-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="web">
          <article-title>Welcome to Medical Subject Headings</article-title>
          <source>National Institutes of Health National Library of Medicine</source>
          <access-date>2026-01-16</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.nlm.nih.gov/mesh/meshhome.html">https://www.nlm.nih.gov/mesh/meshhome.html</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="web">
          <source>Rayyan</source>
          <access-date>2025-12-19</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://new.rayyan.ai/">https://new.rayyan.ai/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Rivera</surname>
              <given-names>SC</given-names>
            </name>
            <name name-style="western">
              <surname>Moher</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Calvert</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Denniston</surname>
              <given-names>AK</given-names>
            </name>
            <collab>SPIRIT-AI and CONSORT-AI Working Group</collab>
          </person-group>
          <article-title>Reporting guidelines for clinical trial reports for interventions involving artificial intelligence: the CONSORT-AI extension</article-title>
          <source>BMJ</source>
          <year>2020</year>
          <month>09</month>
          <day>09</day>
          <volume>370</volume>
          <fpage>m3164</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.bmj.com/lookup/pmidlookup?view=long&#38;pmid=32909959"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmj.m3164</pub-id>
          <pub-id pub-id-type="medline">32909959</pub-id>
          <pub-id pub-id-type="pmcid">PMC7490784</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mayring</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Qualitative content analysis: theoretical foundation, basic procedures and software solution</article-title>
          <source>Social Science Open Access Repository</source>
          <year>2014</year>
          <access-date>2026-07-31</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.ssoar.info/ssoar/handle/document/39517">https://www.ssoar.info/ssoar/handle/document/39517</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="web">
          <article-title>Artificial Intelligence Risk Management Framework (AI RMF 1.0)</article-title>
          <source>National Institute of Standards and Technology</source>
          <year>2023</year>
          <month>01</month>
          <access-date>2026-08-04</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.100-1.pdf">https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.100-1.pdf</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Page</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>McKenzie</surname>
              <given-names>JE</given-names>
            </name>
            <name name-style="western">
              <surname>Bossuyt</surname>
              <given-names>PM</given-names>
            </name>
            <name name-style="western">
              <surname>Boutron</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Hoffmann</surname>
              <given-names>TC</given-names>
            </name>
            <name name-style="western">
              <surname>Mulrow</surname>
              <given-names>CD</given-names>
            </name>
            <name name-style="western">
              <surname>Shamseer</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Tetzlaff</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Akl</surname>
              <given-names>EA</given-names>
            </name>
            <name name-style="western">
              <surname>Brennan</surname>
              <given-names>SE</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Glanville</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Grimshaw</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Hróbjartsson</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lalu</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Loder</surname>
              <given-names>EW</given-names>
            </name>
            <name name-style="western">
              <surname>Mayo-Wilson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>McDonald</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>McGuinness</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Stewart</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Thomas</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tricco</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Welch</surname>
              <given-names>VA</given-names>
            </name>
            <name name-style="western">
              <surname>Whiting</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Moher</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>The PRISMA 2020 statement: an updated guideline for reporting systematic reviews</article-title>
          <source>BMJ</source>
          <year>2021</year>
          <month>03</month>
          <day>29</day>
          <volume>372</volume>
          <fpage>n71</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.bmj.com/lookup/pmidlookup?view=long&#38;pmid=33782057"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmj.n71</pub-id>
          <pub-id pub-id-type="medline">33782057</pub-id>
          <pub-id pub-id-type="pmcid">PMC8005924</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="web">
          <source>DeepL</source>
          <access-date>2026-08-06</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.deepl.com/de">https://www.deepl.com/de</ext-link>
          </comment>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
