<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.0" xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">ResProt</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Res Protoc</journal-id>
      <journal-title>JMIR Research Protocols</journal-title>
      <issn pub-type="epub">1929-0748</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v15i1e99807</article-id>
      <article-id pub-id-type="pmid"/>
      <article-id pub-id-type="doi">10.2196/99807</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Protocol</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Protocol</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Large Language Models for Clinical Data Extraction in Workplace Injury Rehabilitation: Protocol for a Retrospective Pilot Study of Accuracy, Fairness, and Methodological Considerations in Workers’ Compensation Medical Chart Review</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Sarvestan</surname>
            <given-names>Javad</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Toben</surname>
            <given-names>Daan</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Shah</surname>
            <given-names>Armaan Rehman</given-names>
          </name>
          <degrees>BSc</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0007-4748-0071</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Gorczyca Abel</surname>
            <given-names>Barbara</given-names>
          </name>
          <degrees>BSc, BScPT, MHM</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-0485-4654</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Komeili</surname>
            <given-names>Majid</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-4695-3072</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Gohar</surname>
            <given-names>Basem</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff4" ref-type="aff">4</xref>
          <xref rid="aff5" ref-type="aff">5</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-8131-1190</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Nowrouzi-Kia</surname>
            <given-names>Behdin</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Department of Occupational Science and Occupational Therapy</institution>
            <institution>Temerty Faculty of Medicine</institution>
            <institution>University of Toronto</institution>
            <addr-line>500 University Avenue</addr-line>
            <addr-line>Toronto, ON, M5G 1V7</addr-line>
            <country>Canada</country>
            <phone>1 416 946 3249</phone>
            <email>behdin.nowrouzi.kia@utoronto.ca</email>
          </address>
          <xref rid="aff5" ref-type="aff">5</xref>
          <xref rid="aff6" ref-type="aff">6</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5586-4282</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Department of Occupational Science and Occupational Therapy</institution>
        <institution>Temerty Faculty of Medicine</institution>
        <institution>University of Toronto</institution>
        <addr-line>Toronto, ON</addr-line>
        <country>Canada</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Institute for Better Health</institution>
        <institution>Trillium Health Partners</institution>
        <addr-line>Mississauga, ON</addr-line>
        <country>Canada</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>School of Computer Science</institution>
        <institution>Carleton University</institution>
        <addr-line>Ottawa, ON</addr-line>
        <country>Canada</country>
      </aff>
      <aff id="aff4">
        <label>4</label>
        <institution>Department of Population Medicine</institution>
        <institution>University of Guelph</institution>
        <addr-line>Guelph, ON</addr-line>
        <country>Canada</country>
      </aff>
      <aff id="aff5">
        <label>5</label>
        <institution>Centre for Research in Occupational Safety and Health</institution>
        <institution>Laurentian University</institution>
        <addr-line>Sudbury, ON</addr-line>
        <country>Canada</country>
      </aff>
      <aff id="aff6">
        <label>6</label>
        <institution>Krembil Research Institute-University Health Network</institution>
        <institution>Toronto Western Hospital</institution>
        <addr-line>Toronto, ON</addr-line>
        <country>Canada</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Behdin Nowrouzi-Kia <email>behdin.nowrouzi.kia@utoronto.ca</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>22</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>15</volume>
      <elocation-id>e99807</elocation-id>
      <history>
        <date date-type="received">
          <day>29</day>
          <month>4</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>24</day>
          <month>7</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>18</day>
          <month>8</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>31</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Armaan Rehman Shah, Barbara Gorczyca Abel, Majid Komeili, Basem Gohar, Behdin Nowrouzi-Kia. Originally published in JMIR Research Protocols (https://www.researchprotocols.org), 22.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Research Protocols, is properly cited. The complete bibliographic information, a link to the original publication on https://www.researchprotocols.org, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.researchprotocols.org/2026/1/e99807" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Electronic medical charts within Workplace Safety and Insurance Board (WSIB) specialty programs contain rich but unstructured clinical and sociodemographic data essential for injury classification, treatment planning, and compensation decisions. Manual chart review is time-consuming, inconsistent, and susceptible to reviewer bias. Large language models (LLMs) extract complex information from unstructured clinical text, yet their application to workplace injury rehabilitation remains unexplored.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study will evaluate the accuracy, fairness, and methodological implications of using LLMs to extract clinical and rehabilitative information from WSIB medical charts at Trillium Health Partners (THP), Canada. We will assess model performance, examine algorithmic bias across demographic subgroups, and explore ethical implications for clinical decision-making and workplace compensation.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We will conduct a retrospective review of 50 medical charts from the WSIB Back and Neck specialty program at THP, spanning January 2018 to December 2024. General-purpose models (Qwen3-VL-8B and InternVL3.5-8B) and domain-specific biomedical models (MedGemma-27B and LLaMA-3-Meditron-8B) will be evaluated off the shelf using zero-shot and few-shot prompting; the biomedical models will also be fine-tuned. Charts will be partitioned at the chart level into development, training, and held-out test sets, preventing leakage across purposes. Ground truth will be established by independent human annotation of all 50 charts, with interannotator agreement quantified using Cohen κ. The primary outcome is the macroaveraged <italic>F</italic><sub>1</sub>-score across categorical variables on the held-out test set; secondary outcomes are the per variable <italic>F</italic><sub>1</sub>-score, mean absolute error and root mean squared error for continuous variables, and span-level <italic>F</italic><sub>1</sub>-score. Progression to the larger study requires a macroaveraged <italic>F</italic><sub>1</sub>-score of at least 0.80, a pragmatic feasibility criterion; fairness and continuous-variable results are supporting outcomes. Algorithmic fairness will be examined using demographic parity and equalized odds across subgroups. A secure hybrid architecture will keep all identifiable personal health information within the THP infrastructure, with cloud compute restricted to transient processing.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>This study was funded by the Data Sciences Institute at the University of Toronto in April 2025; the award supported protocol development and ended on April 30, 2026. As of August 12, 2026, neither had a research ethics application been submitted nor had any chart been accessed. Applications to the THP and University of Toronto research ethics boards are anticipated in late 2026, and no data will be retrieved or processed before approval from both boards. Contingent on approval and further funding, data collection is anticipated through 2027, analysis in late 2027 to early 2028, and results submitted for publication in 2028.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>This protocol describes the first systematic evaluation of LLM-based data extraction applied to workplace injury medical charts. Findings will inform best practices for the responsible, reproducible, and equitable integration of these tools into occupational health and workers’ compensation decision-making.</p>
        </sec>
        <sec sec-type="registered-report">
          <title>International Registered Report Identifier (IRRID)</title>
          <p>PRR1-10.2196/99807</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>artificial intelligence</kwd>
        <kwd>AI</kwd>
        <kwd>large language model</kwd>
        <kwd>LLM</kwd>
        <kwd>natural language processing</kwd>
        <kwd>occupational rehabilitation</kwd>
        <kwd>workplace injury</kwd>
        <kwd>workers’ compensation</kwd>
        <kwd>algorithmic fairness</kwd>
        <kwd>return to work</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Workplace injuries impose a substantial burden on individuals, employers, and health care systems worldwide. Successful return-to-work (RTW) outcomes are essential for minimizing long-term physical, mental, and economic consequences for injured workers [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Clinical decision-making in workplace injury rehabilitation depends heavily on accurate assessment of injury severity, functional limitations, treatment pathways, and prognostic factors documented in electronic medical charts, and the effectiveness of rehabilitation and RTW interventions rests on how well this clinical information is captured and applied [<xref ref-type="bibr" rid="ref3">3</xref>]. Within Ontario, Canada, the Workplace Safety and Insurance Board (WSIB) administers specialty recovery programs that provide targeted rehabilitation for workers with approved workplace injury claims [<xref ref-type="bibr" rid="ref4">4</xref>].</p>
      <p>Electronic health records (EHRs) within WSIB specialty programs contain rich clinical and sociodemographic information, including diagnoses, injury mechanisms, occupational duties, treatment plans, and RTW barriers. However, extracting structured data from these records is a resource-intensive process that traditionally relies on manual chart reviews by trained clinicians or research personnel. Manual review is time-consuming, susceptible to human bias and inconsistency, and poorly scalable to large datasets [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. These limitations constrain researchers’ and decision-makers’ ability to leverage clinical data for quality improvement, outcome evaluation, and equitable service delivery.</p>
      <p>Recent advances in artificial intelligence (AI), particularly large language models (LLMs), offer a promising approach to automating the extraction of structured information from unstructured clinical text [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Transformer-based models, such as Bidirectional Encoder Representations from Transformers for Biomedical Text Mining (BioBERT), Clinical Bidirectional Encoder Representations from Transformers (ClinicalBERT), and Medical Bidirectional Encoder Representations from Transformers (Med-BERT), have demonstrated effectiveness in extracting diagnoses, medications, and symptoms from EHRs [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref11">11</xref>]. Newer general-purpose LLMs, including Qwen3-VL-8B [<xref ref-type="bibr" rid="ref12">12</xref>] and InternVL3.5-8B [<xref ref-type="bibr" rid="ref13">13</xref>], have shown improved contextual reasoning and adaptability to previously unseen clinical tasks through zero-shot and few-shot learning, and general-purpose models of this class have performed competitively on medical reasoning benchmarks [<xref ref-type="bibr" rid="ref14">14</xref>]. More recent domain-specific models, such as MedGemma-27B [<xref ref-type="bibr" rid="ref15">15</xref>] and LLaMA-3-Meditron-8B [<xref ref-type="bibr" rid="ref16">16</xref>], the latter continuing a line of medical pretraining that began with MEDITRON [<xref ref-type="bibr" rid="ref17">17</xref>] on the LLaMA (Large Language Model Meta AI) architecture [<xref ref-type="bibr" rid="ref18">18</xref>], have been developed with a focus on biomedical and clinical reasoning, enabling improved performance on tasks involving diagnosis extraction, symptom identification, and interpretation of EHRs. Bannett et al [<xref ref-type="bibr" rid="ref19">19</xref>] demonstrated that LLMs can identify clinically relevant information in pediatric medication monitoring records, with accuracy comparable to clinician review, supporting the potential for hybrid human-AI workflows in chart review contexts.</p>
      <p>Despite growing evidence supporting the application of LLMs in clinical data extraction across various medical domains, no existing study has systematically evaluated how LLMs perform in reviewing WSIB-related patient records. Workplace injury medical charts pose unique challenges not addressed in prior research, including documentation of occupational duties and demands, injury mechanisms, RTW barriers, functional capacity evaluations, and multidisciplinary rehabilitation plans. Furthermore, the potential for algorithmic bias in AI-driven extraction from occupational health records, which directly affects worker compensation and access to rehabilitation, has not been critically examined [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>].</p>
      <p>The evidence base for clinical extraction has grown quickly but remains concentrated in a small number of settings. A systematic review of generative LLM applications to EHR data found that evaluation work clusters in a limited set of clinical fields and that reporting of evaluation design is frequently incomplete [<xref ref-type="bibr" rid="ref22">22</xref>]. A scoping review of oncology extraction reached a similar conclusion, identifying rapid growth alongside inconsistent definitions of accuracy and limited external validation [<xref ref-type="bibr" rid="ref23">23</xref>]. Locally deployed open-weight models have been shown to extract structured clinical features from free-text records with high sensitivity and specificity without transmitting text beyond institutional infrastructure [<xref ref-type="bibr" rid="ref24">24</xref>], and privacy-preserving deployment strategies of this kind are now regarded as a precondition for secondary use of health records with generative models [<xref ref-type="bibr" rid="ref25">25</xref>]. Occupational rehabilitation is absent from this literature [<xref ref-type="bibr" rid="ref26">26</xref>]. Machine learning (ML) applied to RTW has focused on outcome prediction from structured variables rather than on extraction from clinical narrative [<xref ref-type="bibr" rid="ref27">27</xref>], leaving the documentation itself unexamined as a data source.</p>
      <p>The absence of empirical research on LLMs in workplace injury assessment represents a significant gap, given the high societal stakes of decisions informed by chart review. Inaccurate or biased data extraction could influence injury classification, treatment recommendations, and compensation outcomes, with disproportionate impacts on marginalized worker populations [<xref ref-type="bibr" rid="ref28">28</xref>].</p>
      <p>This study aims to address the identified knowledge gap through a pilot investigation with the following objectives: (1) assess the accuracy, sensitivity, and specificity of LLMs in extracting clinical-rehabilitative outcomes and sociodemographic data from WSIB medical charts at Trillium Health Partners (THP), Canada; (2) analyze demographic differences in AI-driven classifications of workplace injuries, particularly for marginalized worker populations; (3) examine the ethical and methodological implications of using LLMs in health care decision-making and workplace compensation claims processing; and (4) develop recommendations for integrating AI in clinical-rehabilitative programs to improve quality of care, while ensuring ethical rigor.</p>
      <p>This protocol describes the planned methodology for a pilot study that will serve as a proof-of-concept investigation, generating foundational evidence for responsible, reproducible, and equitable AI integration in occupational health clinical workflows.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Study Design</title>
        <p>This study will use a retrospective chart review design to evaluate the feasibility and performance of LLM-based clinical data extraction from workplace injury medical records. The study design will follow a proof-of-concept framework, beginning with a pilot sample to establish baseline model performance before proceeding to the full dataset. The study is registered with the Data Sciences Institute at the University of Toronto (grant number DSI-CIDSY4R2P01).</p>
      </sec>
      <sec>
        <title>Setting and Data Source</title>
        <p>Data will be obtained from electronic medical charts within the WSIB specialty programs at THP, a large hospital system in the Greater Toronto Area, Ontario, Canada. THP offers seven specialty recovery programs for workplace injuries, funded by the WSIB: Back and Neck, Shoulder and Elbow, Hand and Wrist, Neurology, Foot and Ankle, Hip and Knee, and COVID-19 Assessment Program (CAP). Each program focuses on injury-specific rehabilitation to facilitate rapid recovery and RTW. Previous analyses have demonstrated that participation in four of the seven programs (Back and Neck, Shoulder and Elbow, Hand and Wrist, and Neurology) is effective in reducing depression and anxiety, although these analyses did not address outcomes in the context of RTW [<xref ref-type="bibr" rid="ref4">4</xref>]. <xref ref-type="table" rid="table1">Table 1</xref> provides an overview of the WSIB specialty programs at THP.</p>
        <p>For this pilot study, we will begin with 50 retrospective medical charts from the WSIB Back and Neck specialty program, selected because records in this program are typically more concise, thereby reducing the time required for initial chart review and model evaluation. Records will span from January 2018 to December 2024. The broader dataset comprises approximately 1738 records across all WSIB specialty programs and will be used in subsequent phases pending pilot results.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>WSIB<sup>a</sup> specialty programs at THP<sup>b</sup>.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="240"/>
            <col width="760"/>
            <thead>
              <tr valign="top">
                <td>Program</td>
                <td>Target population</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Back and Neck</td>
                <td>Workers with injuries to the cervical, thoracic, and lumbar spine</td>
              </tr>
              <tr valign="top">
                <td>Shoulder and Elbow</td>
                <td>Workers with injuries to the upper extremities, including the shoulder and elbow</td>
              </tr>
              <tr valign="top">
                <td>Hand and Wrist</td>
                <td>Workers with injuries to the hand or wrist</td>
              </tr>
              <tr valign="top">
                <td>Neurology</td>
                <td>Workers with head injuries, including mild traumatic brain injury and concussion</td>
              </tr>
              <tr valign="top">
                <td>Foot and Ankle</td>
                <td>Workers with injuries to the foot and ankle</td>
              </tr>
              <tr valign="top">
                <td>Hip and Knee</td>
                <td>Workers with injuries to the lower extremities, including the hip and knee</td>
              </tr>
              <tr valign="top">
                <td>CAP<sup>c</sup></td>
                <td>Workers with WSIB-approved COVID-19 claims post-COVID infection</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>WSIB: Workplace Safety and Insurance Board.</p>
            </fn>
            <fn id="table1fn2">
              <p><sup>b</sup>THP: Trillium Health Partners.</p>
            </fn>
            <fn id="table1fn3">
              <p><sup>c</sup>CAP: COVID-19 Assessment Program.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Variables and Data Extraction</title>
        <p>A standardized extraction schema will be developed to capture clinical, rehabilitative, and sociodemographic variables from each medical chart. <xref ref-type="table" rid="table2">Table 2</xref> presents the categories of variables targeted for extraction.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Categories of variables targeted for extraction from WSIB<sup>a</sup> medical charts.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="260"/>
            <col width="740"/>
            <thead>
              <tr valign="top">
                <td>Category</td>
                <td>Variables</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Sociodemographic</td>
                <td>Age, sex, gender, language, occupation, industry, employment status</td>
              </tr>
              <tr valign="top">
                <td>Injury characteristics</td>
                <td>Injury type, mechanism of injury, body region, date of injury, diagnosis codes</td>
              </tr>
              <tr valign="top">
                <td>Clinical assessment</td>
                <td>Pain severity, functional limitations, range of motion, cognitive status, comorbidities</td>
              </tr>
              <tr valign="top">
                <td>Treatment and rehabilitation</td>
                <td>Treatment modalities, rehabilitation goals, number of sessions, duration of program</td>
              </tr>
              <tr valign="top">
                <td>RTW<sup>b</sup> outcomes</td>
                <td>RTW status at discharge, time to RTW, work modifications, barriers to RTW</td>
              </tr>
              <tr valign="top">
                <td>Mental health</td>
                <td>Depression and anxiety screening scores (baseline and discharge), psychological referrals</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>WSIB: Workplace Safety and Insurance Board.</p>
            </fn>
            <fn id="table2fn2">
              <p><sup>b</sup>RTW: return to work.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Chart Structure and Data Representation</title>
        <p>Charts in the Back and Neck specialty program combine narrative clinical documentation with a structured discharge outcome record. The narrative component includes WSIB referral documentation, the initial assessment, progress notes recorded across the rehabilitation episode, and the discharge report, and it is the principal source for variables such as occupation and industry, mechanism of injury, functional limitations described in clinical language, and treatment content, where a single variable may be described in more than one section of a chart and in different terms by different clinicians. The structured component is a standardized discharge data template completed for each patient, which records administrative and clinical information in coded fields: program and treatment stream; treatment start and end dates; total number of visits; dates of injury and of surgery, where applicable; primary area of injury; physician diagnoses selected from region-specific categories; and confounding factors affecting length of stay, such as depression or anxiety, posttraumatic stress disorder, or financial and transportation barriers, selected from a defined list.</p>
        <p>RTW status is likewise captured in coded form at baseline and at discharge, recorded as working full-time or part-time with regular or modified duties or as one of several defined not-working categories, together with a staged rating of RTW readiness. Standardized outcome instruments are administered at program entry and at discharge and recorded in the template as numeric scores. The “depression and anxiety screening scores” listed in <xref ref-type="table" rid="table2">Table 2</xref> are the anxiety and depression subscale scores of the Hospital Anxiety and Depression Scale, captured at baseline and discharge. Pain severity is recorded on the Numeric Pain Rating Scale, condition-specific function on the Oswestry Disability Index and the Neck Disability Index for this program, fear of movement on the Tampa Scale of Kinesiophobia, and health-related quality of life on the EQ-5D-3L, alongside measured functional tests, including timed walk, lifting, carrying, and stair tasks. The variables in <xref ref-type="table" rid="table2">Table 2</xref> therefore arrive in mixed representation: some are read directly from coded template fields or instrument scores, others must be extracted from free narrative text, and some, such as RTW barriers, may appear in both and in different terms.</p>
        <p>Human annotators and language models will work from the same source documents and the same extraction schema, so any difference in performance reflects a difference in extraction rather than a difference in the information available. For each variable in <xref ref-type="table" rid="table2">Table 2</xref>, the schema will specify the permitted value set, the document sections in which the variable is expected to appear, and the rule to apply when a chart contains repeated or conflicting entries, for example, by taking the most recent value recorded before discharge. A variable that is absent from a chart will be recorded explicitly as not documented rather than left blank, so failure to extract a value that was recorded can be distinguished from correct recognition that no value exists. The completed schema, including field definitions and coding rules, will be released as supplementary material.</p>
      </sec>
      <sec>
        <title>Eligibility Criteria</title>
        <p>Eligible records are those of adults injured at work and referred to WSIB specialty programs with board-approved claims. No maximum age will be imposed, although the sample will primarily consist of individuals younger than 65 years of age, given the nature of employment-based injuries. The study will use preexisting clinical records created in the course of care rather than data collected for research purposes. No participant will be contacted, and no clinical intervention will be introduced.</p>
      </sec>
      <sec>
        <title>AI Model Selection and Architecture</title>
        <p>We will evaluate four models in their off-the-shelf (no fine-tuning) form: two general-purpose models (Qwen3-VL-8B and InternVL3.5-8B) and two domain-specific biomedical models (MedGemma-27B and LLaMA-3-Meditron-8B). The general-purpose models are included because their broader pretraining may enable stronger performance on nonclinical or semistructured variables, such as sociodemographic information, where domain-specific knowledge is less critical. In addition, the two biomedical models (MedGemma-27B and LLaMA-3-Meditron-8B) will be further evaluated under a fine-tuning setting using the annotated pilot dataset to assess potential gains in extracting clinically nuanced variables. All models will be prompted with structured queries (eg, “What is the patient’s primary diagnosis?”) to extract predefined variables across sociodemographic, clinical, treatment, and mental health domains. <xref ref-type="table" rid="table3">Table 3</xref> summarizes the models, their characteristics, and intended roles in this study.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>LLMs<sup>a</sup> evaluated in this study.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="280"/>
            <col width="160"/>
            <col width="220"/>
            <col width="340"/>
            <thead>
              <tr valign="top">
                <td>Model</td>
                <td>Parameters</td>
                <td>Domain</td>
                <td>Approach</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Qwen3-VL-8B</td>
                <td>~9B</td>
                <td>General purpose</td>
                <td>Without fine-tuning</td>
              </tr>
              <tr valign="top">
                <td>InternVL3.5-8B</td>
                <td>~9B</td>
                <td>General purpose</td>
                <td>Without fine-tuning</td>
              </tr>
              <tr valign="top">
                <td>MedGemma-27B</td>
                <td>27B</td>
                <td>Medical</td>
                <td>With and without fine-tuning</td>
              </tr>
              <tr valign="top">
                <td>LLaMA-3-Meditron-8B<sup>b</sup></td>
                <td>8B</td>
                <td>Medical</td>
                <td>With and without fine-tuning</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>LLM: large language model.</p>
            </fn>
            <fn id="table3fn2">
              <p><sup>b</sup>LLaMA: Large Language Model Meta AI.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Human Annotation and Ground Truth</title>
        <p>To establish a reference standard for evaluating model performance, human annotation will be performed on the full pilot sample of 50 medical charts. Three independent annotators with clinical or research expertise in occupational health will review each chart using the standardized extraction schema. Annotators will extract the same set of variables targeted by the AI models. Interannotator agreement will be assessed using Cohen κ to evaluate consistency across annotators and to identify variables that require additional clarification or standardization of extraction criteria [<xref ref-type="bibr" rid="ref29">29</xref>]. Discrepancies will be resolved through consensus discussion. The consensus annotations will serve as the ground truth against which all model outputs will be compared.</p>
      </sec>
      <sec>
        <title>Data Partitioning</title>
        <p>Prompt development, fine-tuning, and final evaluation will be carried out on separate, nonoverlapping sets of charts. Using the same records for all three purposes would allow information from the evaluation data to influence model configuration and would yield performance estimates that are optimistically biased and unlikely to reproduce on unseen records. This failure mode is well documented across applied ML and is a recognized source of irreproducible findings [<xref ref-type="bibr" rid="ref30">30</xref>], and its avoidance is an explicit requirement of current consensus reporting standards for ML-based science [<xref ref-type="bibr" rid="ref31">31</xref>].</p>
        <p>The 50 charts will be partitioned once, before any model is run, into a development set of 10 (20%) charts, a training set of 25 (50%) charts, and a held-out test set of 15 (30%) charts. Partitioning will be performed at the chart level so that no record contributes to more than one set and will be stratified by sex and by broad age band to reduce the chance that a demographic subgroup is absent from the test set. The allocation will be generated with a fixed random seed, which will be reported so that the partition can be reproduced.</p>
        <p>The development set will be the only data used to design and revise extraction prompts and to select few-shot exemplars. The training set will be used exclusively for supervised fine-tuning of the two biomedical models. The held-out test set will not be inspected during prompt development or fine-tuning and will be used once, at the end of the study, to produce the reported performance and fairness estimates for all four models under all three extraction approaches. Where fine-tuning requires a validation signal for early stopping or hyperparameter selection, that signal will be obtained by cross-validation within the training set and never from the test set. Should any test chart be examined for an unanticipated reason, the event and the charts involved will be reported alongside the results.</p>
        <p>Human annotation will cover all 50 charts, so consensus ground truth will be available for every partition, and annotators will not be informed of partition membership. Reporting will follow the TRIPOD-LLM statement for studies developing or evaluating LLMs in health care [<xref ref-type="bibr" rid="ref32">32</xref>], together with the TRIPOD+AI (Transparent Reporting of a Multivariable Prediction Model for individual Prognosis or Diagnosis+AI) statement for prediction model reporting [<xref ref-type="bibr" rid="ref33">33</xref>].</p>
      </sec>
      <sec>
        <title>Model Performance Evaluation</title>
        <p>Model performance will be evaluated by comparing AI-extracted data against the human-annotated ground truth. Performance metrics will be selected based on the type of variable being extracted. For binary variables, we will report sensitivity (true-positive rate), specificity (true-negative rate), the positive predictive value (precision), and the <italic>F</italic><sub>1</sub>-score. For multiclass classification tasks, macro- and microaveraged <italic>F</italic><sub>1</sub>-scores and confusion matrices will be used. For continuous variables, including numerical clinical measures, we will report the mean absolute error (MAE) and the root mean squared error (RMSE). For text span and entity extraction tasks, such as diagnosis codes and free-text clinical concepts, we will compute token- and span-level precision, recall, and the <italic>F</italic><sub>1</sub>-score. Following the approach described by Bannett et al [<xref ref-type="bibr" rid="ref19">19</xref>], we will assess model explainability by qualitatively examining extraction errors to identify systematic patterns of misclassification. Performance will be compared across the two tiers (general-purpose vs domain-specific models) and the three extraction approaches (zero-shot, few-shot, and fine-tuned). All figures reported as final performance will be computed on the held-out test set defined earlier. Performance on the development or training partitions, where reported, will be labeled as such and will not be presented as an estimate of generalization performance.</p>
        <p>The primary model performance outcome is the macroaveraged <italic>F</italic><sub>1</sub>-score across the categorical variables listed in <xref ref-type="table" rid="table2">Table 2</xref>, computed on the held-out test set and reported separately for each model and each extraction approach. Macroaveraging is used in preference to microaveraging so that variables with few documented instances contribute equally to the primary outcome rather than being dominated by the most frequently documented categories, which also preserves sensitivity to the subgroup effects examined in the fairness assessment. Secondary outcomes are the per variable <italic>F</italic><sub>1</sub>-score, the MAE and RMSE for continuous variables, the token- and span-level <italic>F</italic><sub>1</sub>-score for narrative extraction, and the fairness metrics reported in <xref ref-type="table" rid="table4">Table 4</xref>.</p>
        <p>As a descriptive benchmark for interpreting model performance, we will also compute a human reference ceiling, defined as the macroaveraged <italic>F</italic><sub>1</sub>-score of each individual annotator against the consensus ground truth, averaged across the three annotators and computed on the same test set and the same variables as the model estimates. This places model and human performance on a common scale and indicates how much of the shortfall from perfect agreement reflects the difficulty of the charts themselves rather than a limitation of the models. It is reported as a context for interpretation and is not a component of the progression criterion.</p>
        <p>Progression to the larger study will be considered justified if the best-performing model achieves a macroaveraged <italic>F</italic><sub>1</sub>-score of at least 0.80 on the held-out test set, without evidence of systematic failure on clinically important variables. This threshold is stated as a pragmatic feasibility criterion for deciding whether to proceed to the larger dataset, not as a validated threshold for acceptable clinical performance or for deployment in compensation decision-making; no such threshold has been established for this task. Evidence of systematic failure will be assessed by inspecting per variable performance for the injury characteristics and RTW outcome variables in <xref ref-type="table" rid="table2">Table 2</xref>, which carry the most direct consequences for compensation and care, and any variable on which extraction fails consistently will be reported and discussed rather than absorbed into the aggregate.</p>
        <p>Continuous-variable performance, narrative extraction performance, and subgroup fairness results will be reported as supporting outcomes rather than as components of the progression criterion. This reflects the precision available from a held-out test set of 15 charts, in which subgroup cells in particular will be small, so that committing in advance to numerical disparity thresholds would imply a precision the pilot cannot support. All primary and supporting estimates will be reported with 95% CIs, and where a CI spans the progression threshold, the result will be reported as indeterminate rather than as meeting or failing it. The fairness results will inform the design and analysis plan of the larger study, in which subgroup sample sizes permit meaningful estimation.</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Fairness metrics for algorithmic bias assessment.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="220"/>
            <col width="400"/>
            <col width="380"/>
            <thead>
              <tr valign="top">
                <td>Metric</td>
                <td>Definition</td>
                <td>Assessment</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Demographic parity</td>
                <td>The proportion of positive predictions should be equal across demographic groups.</td>
                <td>Compare positive prediction rates across subgroups</td>
              </tr>
              <tr valign="top">
                <td>Equalized odds</td>
                <td>True- and false-positive rates should be equal across demographic groups.</td>
                <td>Compare true- and false-positive rates across subgroups for each variable</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Fairness and Bias Assessment</title>
        <p>A critical component of this study will be the assessment of algorithmic fairness in AI-driven data extraction. We will evaluate whether model performance differs systematically across demographic subgroups defined by age, sex, language, occupation type, and industry. Two primary fairness metrics will be applied, as described in <xref ref-type="table" rid="table4">Table 4</xref>. Fairness metrics will be computed on the held-out test set against the same consensus ground truth used for accuracy. Because subgroup sample sizes within a 50-chart pilot will be small, subgroup estimates will be reported with exact CIs and read as preliminary signals for the larger study rather than as evidence for or against the presence of bias. Systematic reviews of medical language models report demographic disparities across a wide range of tasks and find mitigation methods to be comparatively immature [<xref ref-type="bibr" rid="ref34">34</xref>], and closed models have been shown to reproduce racial and gender patterns present in clinical text [<xref ref-type="bibr" rid="ref35">35</xref>], which supports reporting observed disparities transparently rather than only after a mitigation step has been applied.</p>
        <p>In addition, qualitative analysis of extraction errors will be conducted to identify potential sources of harm, including misclassification patterns that may disproportionately affect specific worker populations. This assessment will be informed by clinical expertise and contextualized judgment from team members with experience in occupational health and rehabilitation practice.</p>
      </sec>
      <sec>
        <title>Ethical and Social Analysis</title>
        <p>Beyond quantitative model evaluation, this study will examine the ethical and social implications of AI-assisted chart review in occupational health. This component will analyze three data sources generated by the study itself: the catalog of extraction errors produced on the held-out test set, the subgroup performance differences produced by the fairness assessment, and the recorded disagreements between annotators, which identify the variables whose meaning is contested even among trained human reviewers.</p>
        <p>Analysis will proceed in two stages. First, each error type observed in the test set will be mapped to the care or compensation decision it could plausibly affect. An omitted RTW restriction bears on workplace accommodation, whereas a misclassified injury mechanism bears on claim adjudication. This mapping will produce a structured risk register recording, for each variable in <xref ref-type="table" rid="table2">Table 2</xref>, the observed error rate, the direction of error, the decision affected, and the party who would bear the consequence. Second, the research team will hold structured analysis sessions in which occupational therapy, clinical, and data science members will review the register together and assess where automated extraction would shift interpretive authority away from clinicians and injured workers and toward the operators of the extraction system. These sessions will be documented, and points of disagreement within the team will be recorded rather than resolved into a single position.</p>
        <p>The outputs of this component will be the completed risk register, a list of variables identified as unsuitable for automated extraction without human verification, and a written account of the conditions under which the research team judges LLM-assisted chart review to be defensible in a compensation context. These outputs are intended to inform governance requirements for the larger study rather than to serve as a general endorsement of the method.</p>
      </sec>
      <sec>
        <title>Data Security and Privacy Architecture</title>
        <p>Given that this study involves identifiable personal health information (PHI), a secure hybrid architecture will be implemented to enable graphics processing unit (GPU)–accelerated LLM processing, while ensuring that PHI never leaves hospital custody. Two categories of data are distinguished throughout this protocol: (1) identifiable PHI, comprising the original clinical records, which remains within THP at all times, and (2) deidentified health data, that is, the study dataset produced after deidentification, which is the only data transmitted to cloud compute resources. The architecture will separate data custody from compute execution: all identifiable PHI will be retained within THP’s on-premises secure sandbox environment (MCVHDSMAN90), while Amazon Web Services (AWS) cloud resources will be used exclusively as transient compute layers. <xref rid="figure1" ref-type="fig">Figure 1</xref> presents an overview of this architecture. Key architectural features include the following:</p>
        <list list-type="bullet">
          <list-item>
            <p>First, a REST API server (EC2 t3.medium) will operate in a private subnet with no public IP address and will serve as the controlled ingress and egress point for all data transmission.</p>
          </list-item>
          <list-item>
            <p>Second, GPU-enabled LLM inference nodes (EC2 g5.2xlarge or g6e.2xlarge) will perform model fine-tuning and inference in private subnets accessible only through AWS PrivateLink.</p>
          </list-item>
          <list-item>
            <p>Third, all data transmitted between THP and AWS will be encrypted using Transport Layer Security (TLS) 1.2 or higher with mutual authentication, routed through a client-to-site virtual private network (VPN).</p>
          </list-item>
          <list-item>
            <p>Fourth, deidentified health data will be decrypted only in memory during processing on inference nodes and will be immediately re-encrypted before return transmission. No health data will be persisted in AWS storage or logs at any point.</p>
          </list-item>
          <list-item>
            <p>Fifth, all encryption keys will be generated and managed exclusively within the THP infrastructure.</p>
          </list-item>
          <list-item>
            <p>Sixth, administrative access to cloud resources will be managed through AWS Systems Manager (SSM) Session Manager, eliminating Secure Shell (SSH) access.</p>
          </list-item>
          <list-item>
            <p>Seventh, all activity will be logged through AWS Control Tower and forwarded to THP’s security information and event management system (Azure Sentinel) for real-time monitoring.</p>
          </list-item>
        </list>
        <p>This architecture will ensure compliance with THP enterprise governance requirements and Canadian privacy legislation, while enabling the computational resources necessary for LLM fine-tuning and inference workloads.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Secure hybrid cloud architecture for the transient processing of WSIB data containing PHI. All PHI remains within the THP on-premises secure sandbox (MCVHDSMAN90). Cloud compute resources in AWS are used exclusively for transient, memory-only processing. Data cross the trust boundary via a client-to-site virtual private network with Transport Layer Security 1.2 or higher and mutual authentication. AWS: Amazon Web Services; PHI: personal health information; SSH: Secure Shell; SSM: Systems Manager; THP: Trillium Health Partners; TLS: Transport Layer Security; VPN: virtual private network; WSIB: Workplace Safety and Insurance Board.</p>
          </caption>
          <graphic xlink:href="resprot_v15i1e99807_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>This manuscript describes a planned study. No chart has been accessed, transferred, or processed, and no data collection will begin before research ethics approval is in place. Approval will be obtained from the research ethics boards of both THP, which holds custody of the records, and the University of Toronto, prior to any data access at either site. The protocol will not proceed to data collection under any other authorization.</p>
        <p>Consent provided by a worker to participate in a treatment program is not consent to the research use of that worker’s record, and this protocol does not treat it as such. Both applications will therefore request a waiver of the requirement to seek individual consent for secondary use of identifiable information under Article 5.5A of the Tri-Council Policy Statement [<xref ref-type="bibr" rid="ref36">36</xref>]. The grounds for the request are that the records were created for clinical and compensation purposes between 2018 and 2024, that many workers completed their programs several years ago and are no longer in contact with the hospital, that recontacting individuals across a 6-year retrospective window is impracticable, that the research could not reasonably be carried out without the waiver, and that the safeguards described next limit residual risk to participants. Whether these grounds are met is a determination for the research ethics boards, and the study will not proceed if the waiver is refused.</p>
        <p>Records will be identifiable at the point of access, because sociodemographic and clinical detail must be read from the original documents. Deidentification will therefore be performed inside the THP secure environment before any content enters the processing pipeline described earlier. Direct identifiers will be removed and replaced with study codes, the linking file will be held only within hospital infrastructure and will be accessible only to hospital-based members of the team, and no identifiable content will cross the trust boundary at any stage. References elsewhere in this protocol to deidentified data describe the state of the data after this step rather than at the point of retrieval.</p>
        <p>Two distinct retention periods apply, and this distinction matters because the study draws on records maintained for clinical reasons, independent of the research. The deidentified research dataset created for this study will be retained for 5 years, in accordance with THP research data retention policy, after which it will be securely destroyed. The underlying clinical records will remain in the custody of THP and will be retained indefinitely for clinical purposes, as they would be irrespective of this study; the research team will hold no copy of them beyond the research retention period. The file linking study codes to patient identifiers will be destroyed once data collection is complete and annotation has been finalized. Retention periods, storage locations, and destruction procedures will be set out in the ethics applications and followed, as approved.</p>
        <p>Data will be accessed, processed, and stored in accordance with the secure hybrid architecture described earlier, so PHI will always remain under hospital custody. Locally deployed open-weight models were selected, in part, for this reason, since they allow inference without transmitting clinical text to an external model provider [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. The research team includes collaborators from THP clinical practice who will contribute expertise in medical chart annotation and interpretation.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <p>This study was funded in April 2025 by the Data Sciences Institute at the University of Toronto through the Critical Investigation of Data Science Grant program (grant number DSI-CIDSY4R2P01), with a total budget of CAD $10,000 (US $7305) and a duration of 12 months (May 1, 2025, to April 30, 2026). The award supported the formation of the research team and the development of the study plan reported here, which were its stated purposes, and its term has now ended. As of August 12, 2026, neither had a research ethics application been submitted nor had research ethics approval been granted, and no chart data have been accessed, extracted, or analyzed. Applications to the THP and University of Toronto research ethics boards are anticipated in late 2026. Contingent on approval from both boards and on securing funding for the data collection phase, chart retrieval and deidentification, human annotation of all 50 charts, model evaluation on the partitioned dataset, and fairness analysis are anticipated to proceed through 2027, with analysis in late 2027 to early 2028 and results submitted for publication in 2028. These dates are estimates contingent on both conditions being met, and the study will not proceed to data collection until they are.</p>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Considerations</title>
        <p>This protocol describes the first systematic investigation of LLM-based clinical data extraction applied to workplace injury medical charts within a workers’ compensation rehabilitation context. The study will address a critical gap at the intersection of AI in health care, occupational rehabilitation, and algorithmic fairness. By using both general-purpose and domain-specific language models, this study will generate foundational evidence regarding the feasibility, accuracy, and equity implications of automated chart review in occupational health settings.</p>
        <p>The model evaluation strategy compares locally deployed open-weight models under both prompting-based and fine-tuning settings. We will assess the off-the-shelf performance of general-purpose models (Qwen3-VL-8B and InternVL3.5-8B) and biomedical models (MedGemma-27B and LLaMA-3-Meditron-8B), as well as the impact of fine-tuning on the biomedical models. This comparison is particularly relevant for clinical settings where the cost and complexity of model fine-tuning must be weighed against the accessibility and flexibility of prompt-based approaches.</p>
      </sec>
      <sec>
        <title>Comparison With Prior Work</title>
        <p>Although prior studies have demonstrated the utility of LLMs in clinical data extraction across domains such as pediatric medication monitoring [<xref ref-type="bibr" rid="ref19">19</xref>], classification of free-text pathology reports [<xref ref-type="bibr" rid="ref37">37</xref>], and prediction of clinical and operational outcomes from clinical notes [<xref ref-type="bibr" rid="ref38">38</xref>], no previous work has applied these methods to workplace injury rehabilitation records. The WSIB chart review context introduces domain-specific challenges, including highly variable documentation structures across specialty programs, the integration of clinical and occupational information within single records, and the need to extract variables (eg, RTW barriers and functional capacity assessments) that are not well represented in general biomedical training corpora. This study will provide the first empirical data on LLM performance in this context.</p>
        <p>The inclusion of a formal fairness assessment using the demographic parity and equalized odds metrics will also distinguish this study from most existing clinical AI evaluations, which typically focus on aggregate accuracy rather than stratified analyses across demographic groups [<xref ref-type="bibr" rid="ref20">20</xref>]. Given that workplace injury outcomes are influenced by sociodemographic factors, including age, sex, occupation, and industry, equitable model performance is essential for responsible deployment.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>Several limitations should be acknowledged, together with the reasoning behind the design choices that produce them. First, the pilot draws 50 charts from a single specialty program at a single facility, which limits the generalizability of initial findings. Restricting the pilot to the Back and Neck specialty program follows from the annotation burden: consensus ground truth requires three independent annotators to review every chart, and within a 12-month grant of CAD $10,000 (US $7305), the inclusion of a second program would either halve the sample available per program or exhaust the annotation capacity that makes the ground truth credible. Three features of the design limit the cost of this restriction. The extraction schema in <xref ref-type="table" rid="table2">Table 2</xref> contains no variable specific to spinal injury and is intended to transfer unchanged to the remaining six programs. Performance will be reported per variable rather than only in aggregate, so variables that depend on program-specific documentation can be distinguished from those that do not. The subsequent phase will draw on the full dataset of approximately 1738 records across all seven programs, and the pilot is structured to produce the schema, the partitioning procedure, and the annotation protocol that this phase requires.</p>
        <p>Second, the retrospective design means that documented variables are subject to confounding not captured by the extraction schema. Because the objective is to measure extraction accuracy against what clinicians recorded rather than to estimate causal effects on RTW, this limitation constrains the interpretation of any secondary analysis rather than the primary outcome. Third, a consensus ground truth established by three annotators may not capture the full range of interpretive variability present in clinical chart review. Reporting Cohen κ per variable and retaining the record of preconsensus disagreement will make visible those variables on which human agreement is unstable, and model performance on such variables will be interpreted against a ground truth that is itself uncertain rather than treated as error.</p>
        <p>Fourth, the models were not trained on occupational health data, which may limit domain-specific performance. This is the reason the design pairs off-the-shelf evaluation with fine-tuning on the training partition, so the gap attributable to domain mismatch is measured rather than assumed. Fifth, the secure hybrid architecture introduces computational overhead that may affect throughput. This cost is accepted because transmitting clinical text to an external model provider is incompatible with hospital data governance, and processing time will be recorded so that the operational cost of the architecture can be reported alongside accuracy. Sixth, a held-out test set of 15 charts yields small subgroup sample sizes for the fairness analysis. The pilot is designed to detect gross disparities and to establish the measurement procedure rather than to produce precise subgroup estimates, which is a principal reason for proceeding to the larger dataset.</p>
      </sec>
      <sec>
        <title>Conclusion</title>
        <p>This protocol outlines a rigorous and ethically informed approach to evaluating LLM-based data extraction in workplace injury rehabilitation. By combining quantitative performance assessment with fairness analysis and critical ethical inquiry, the study aims to establish a foundational methodology for the responsible integration of AI in occupational health. Findings from this pilot will inform the design of a larger-scale study across all WSIB specialty programs at THP, contribute to the development of ethical guidelines for AI-assisted chart review, and support evidence-based policy recommendations for AI deployment in workers’ compensation clinical workflows. The study directly aligns with the Data Sciences Institute’s mission to ensure that data science methodologies are applied in a reproducible, fair, and ethical manner.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group/>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">AI</term>
          <def>
            <p>artificial intelligence</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">AWS</term>
          <def>
            <p>Amazon Web Services</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">BioBERT</term>
          <def>
            <p>Bidirectional Encoder Representations from Transformers for Biomedical Text Mining</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">CAP</term>
          <def>
            <p>COVID-19 Assessment Program</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">ClinicalBERT</term>
          <def>
            <p>Clinical Bidirectional Encoder Representations from Transformers</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">EHR</term>
          <def>
            <p>electronic health record</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">GAI</term>
          <def>
            <p>generative artificial intelligence</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">GPU</term>
          <def>
            <p>graphics processing unit</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">LLaMA</term>
          <def>
            <p>Large Language Model Meta AI</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb11">MAE</term>
          <def>
            <p>mean absolute error</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb12">Med-BERT</term>
          <def>
            <p>Medical Bidirectional Encoder Representations from Transformers</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb13">ML</term>
          <def>
            <p>machine learning</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb14">PHI</term>
          <def>
            <p>personal health information</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb15">RMSE</term>
          <def>
            <p>root mean squared error</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb16">RTW</term>
          <def>
            <p>return to work</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb17">SSH</term>
          <def>
            <p>Secure Shell</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb18">SSM</term>
          <def>
            <p>Systems Manager</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb19">THP</term>
          <def>
            <p>Trillium Health Partners</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb20">TLS</term>
          <def>
            <p>Transport Layer Security</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb21">TRIPOD+AI</term>
          <def>
            <p>Transparent Reporting of a Multivariable Prediction Model for individual Prognosis or Diagnosis+AI</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb22">VPN</term>
          <def>
            <p>virtual private network</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb23">WSIB</term>
          <def>
            <p>Workplace Safety and Insurance Board</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors thank Laura Rodrigues at Trillium Health Partners for technical guidance on the design of the secure data architecture.</p>
      <p>The authors declare the use of generative artificial intelligence (GAI) in the research and writing process. According to the GAIDeT taxonomy (2025), the following tasks were delegated to GAI tools under full human supervision: literature search and systematization, visualization, text generation, proofreading and editing, and reformatting.</p>
      <p>The GAI tool used was Claude (Anthropic), via claude.ai, 2025-2026. Responsibility for the final manuscript lies entirely with the authors. GAI tools are not listed as authors and do not bear responsibility for the final outcomes. Declaration submitted by Armaan Rehman Shah (on behalf of all authors).</p>
      <p>All AI-assisted output was produced under the authors’ direction and reviewed, verified, and revised by them; the authors accept full responsibility for the manuscript. Every reference was verified by the authors against the original published source. An initial draft of <xref rid="figure1" ref-type="fig">Figure 1</xref> was produced with AI assistance and was subsequently revised manually by the authors, who verified its accuracy. GAI was not used to design the study or to generate, analyze, or interpret data; no study data have been collected at this stage, as this protocol precedes research ethics approval.</p>
      <p>This declaration was prepared using the GAIDeT Declaration Generator, based on the GAIDeT taxonomy [<xref ref-type="bibr" rid="ref39">39</xref>].</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>This study was supported by the Data Sciences Institute at the University of Toronto through the Critical Investigation of Data Science Grant program (grant number DSI-CIDSY4R2P01). The funder had no role in the design of this protocol and will play no role in data collection, analysis, interpretation of findings, or the decision to submit results for publication. The award term ended on April 30, 2026, having supported team formation and protocol development. Funding for the data collection phase is being sought, and the principal investigator is continuing protocol development and research ethics preparation in the interim. Data collection will begin only once funding and approval from both research ethics boards are in place.</p>
      </sec>
    </notes>
    <notes>
      <sec>
        <title>Data Availability</title>
        <p>This study will involve retrospective clinical data from Trillium Health Partners (THP) containing personal health information. Data access will be restricted and governed by THP’s data governance policies and the terms of the research ethics approval. Deidentified aggregate results and analysis code will be made available upon publication in a public repository.</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cullen</surname>
              <given-names>KL</given-names>
            </name>
            <name name-style="western">
              <surname>Irvin</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Collie</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Clay</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Gensby</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Jennings</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Hogg-Johnson</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kristman</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Laberge</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>McKenzie</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Newnam</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Palagyi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Ruseckaite</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Sheppard</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Shourie</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Steenstra</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Van Eerd</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Amick</surname>
              <given-names>BC</given-names>
            </name>
          </person-group>
          <article-title>Effectiveness of workplace interventions in return-to-work for musculoskeletal, pain-related and mental health conditions: an update of the evidence and messages for practitioners</article-title>
          <source>J Occup Rehabil</source>
          <year>2018</year>
          <month>03</month>
          <day>21</day>
          <volume>28</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <lpage>15</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/28224415"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s10926-016-9690-x</pub-id>
          <pub-id pub-id-type="medline">28224415</pub-id>
          <pub-id pub-id-type="pii">10.1007/s10926-016-9690-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC5820404</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Franche</surname>
              <given-names>RL</given-names>
            </name>
            <name name-style="western">
              <surname>Cullen</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Clarke</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Irvin</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Sinclair</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Frank</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Workplace-based return-to-work interventions: a systematic review of the quantitative literature</article-title>
          <source>J Occup Rehabil</source>
          <year>2005</year>
          <month>12</month>
          <volume>15</volume>
          <issue>4</issue>
          <fpage>607</fpage>
          <lpage>631</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://link.springer.com/article/10.1007/s10926-005-8038-8"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s10926-005-8038-8</pub-id>
          <pub-id pub-id-type="medline">16254759</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nowrouzi-Kia</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Garrido</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Gohar</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Yazdani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Chattu</surname>
              <given-names>VK</given-names>
            </name>
            <name name-style="western">
              <surname>Bani-Fatemi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Howe</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Duncan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Riquelme</surname>
              <given-names>MP</given-names>
            </name>
            <name name-style="western">
              <surname>Abdullah</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Jaswal</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lo</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Fayyaz</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Alam</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the effectiveness of return-to-work interventions for individuals with work-related mental health conditions: a systematic review and meta-analysis</article-title>
          <source>Healthcare (Basel)</source>
          <year>2023</year>
          <month>05</month>
          <day>12</day>
          <volume>11</volume>
          <issue>10</issue>
          <fpage>1403</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=healthcare11101403"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/healthcare11101403</pub-id>
          <pub-id pub-id-type="medline">37239689</pub-id>
          <pub-id pub-id-type="pii">healthcare11101403</pub-id>
          <pub-id pub-id-type="pmcid">PMC10217842</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="web">
          <article-title>WSIB specialty programs</article-title>
          <source>Workplace Safety and Insurance Board</source>
          <year>2024</year>
          <access-date>2026-03-15</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.wsib.ca">https://www.wsib.ca</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Garies</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Birtwhistle</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Drummond</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Queenan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Williamson</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>Data resource profile: national electronic medical record data from the Canadian Primary Care Sentinel Surveillance Network (CPCSSN)</article-title>
          <source>Int J Epidemiol</source>
          <year>2017</year>
          <month>08</month>
          <day>01</day>
          <volume>46</volume>
          <issue>4</issue>
          <fpage>1091</fpage>
          <lpage>1092f</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/ije/article-lookup/doi/10.1093/ije/dyw248"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/ije/dyw248</pub-id>
          <pub-id pub-id-type="medline">28338877</pub-id>
          <pub-id pub-id-type="pii">3058732</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vassar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Holzmann</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>The retrospective chart review: important methodological considerations</article-title>
          <source>J Educ Eval Health Prof</source>
          <year>2013</year>
          <month>11</month>
          <day>30</day>
          <volume>10</volume>
          <fpage>12</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/24324853"/>
          </comment>
          <pub-id pub-id-type="doi">10.3352/jeehp.2013.10.12</pub-id>
          <pub-id pub-id-type="medline">24324853</pub-id>
          <pub-id pub-id-type="pii">jeehp-10-12</pub-id>
          <pub-id pub-id-type="pmcid">PMC3853868</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Thirunavukarasu</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DSJ</given-names>
            </name>
            <name name-style="western">
              <surname>Elangovan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gutierrez</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>TF</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DSW</given-names>
            </name>
          </person-group>
          <article-title>Large language models in medicine</article-title>
          <source>Nat Med</source>
          <year>2023</year>
          <month>08</month>
          <volume>29</volume>
          <issue>8</issue>
          <fpage>1930</fpage>
          <lpage>1940</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.nature.com/articles/s41591-023-02448-8#citeas"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id>
          <pub-id pub-id-type="medline">37460753</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-023-02448-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tanwani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Seneviratne</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gamble</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Babiker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schärli</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhery</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tomasev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>08</month>
          <day>12</day>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>180</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yoon</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>So</surname>
              <given-names>CH</given-names>
            </name>
            <name name-style="western">
              <surname>Kang</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title>
          <source>Bioinformatics</source>
          <year>2020</year>
          <month>02</month>
          <day>15</day>
          <volume>36</volume>
          <issue>4</issue>
          <fpage>1234</fpage>
          <lpage>1240</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31501885"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id>
          <pub-id pub-id-type="medline">31501885</pub-id>
          <pub-id pub-id-type="pii">5566506</pub-id>
          <pub-id pub-id-type="pmcid">PMC7703786</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Alsentzer</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Murphy</surname>
              <given-names>JR</given-names>
            </name>
            <name name-style="western">
              <surname>Boag</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Weng</surname>
              <given-names>W-H</given-names>
            </name>
            <name name-style="western">
              <surname>Jindi</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Naumann</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>McDermott</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Publicly available clinical BERT embeddings</article-title>
          <year>2019</year>
          <conf-name>2nd Clinical Natural Language Processing Workshop</conf-name>
          <conf-date>June 7, 2019</conf-date>
          <conf-loc>Minneapolis, MN</conf-loc>
          <fpage>72</fpage>
          <lpage>78</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://aclanthology.org/W19-1909.pdf"/>
          </comment>
          <pub-id pub-id-type="doi">10.18653/v1/W19-1909</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rasmy</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Xiang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Tao</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zhi</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Med-BERT: pretrained contextualized embeddings on large-scale structured electronic health records for disease prediction</article-title>
          <source>NPJ Digit Med</source>
          <year>2021</year>
          <month>05</month>
          <day>20</day>
          <volume>4</volume>
          <issue>1</issue>
          <fpage>86</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-021-00455-y"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-021-00455-y</pub-id>
          <pub-id pub-id-type="medline">34017034</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-021-00455-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC8137882</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cai</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Deng</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ding</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Ge</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Ge</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hui</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Qwen3-VL technical report</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on November 26, 2025</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2511.21631">https://arxiv.org/abs/2511.21631</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Gu</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Pu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Cui</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Jing</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Shao</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>InternVL3.5: advancing open-source multimodal models in versatility, reasoning, and efficiency</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on August 25, 2025</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2508.18265">https://arxiv.org/abs/2508.18265</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nori</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>King</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>McKinney</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Carignan</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Horvitz</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Capabilities of GPT-4 on medical challenge problems</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on March 20, 2023</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2303.13375">https://arxiv.org/abs/2303.13375</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sellergren</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kazemzadeh</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jaroensri</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Kiraly</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Traverse</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kohlberger</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jamil</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hughes</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lau</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Mahvar</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Yatziv</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Sterling</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Baby</surname>
              <given-names>SA</given-names>
            </name>
          </person-group>
          <article-title>MedGemma technical report</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on July 7, 2025</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2507.05201">https://arxiv.org/abs/2507.05201</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sallinen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Solergibert</surname>
              <given-names>A-J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Boyé</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Dupont-Roc</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Theimer-Lienhard</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Boisson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Bernath</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Hadhri</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Rabbani</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Brokowski</surname>
              <given-names>T</given-names>
            </name>
            <collab>Meditron Medical Doctor Working Group</collab>
            <name name-style="western">
              <surname>Rudner</surname>
              <given-names>TGJ</given-names>
            </name>
            <name name-style="western">
              <surname>Hartley</surname>
              <given-names>M-A</given-names>
            </name>
          </person-group>
          <article-title>Llama-3-Meditron: an open-weight suite of medical LLMs based on Llama-3.1</article-title>
          <year>2025</year>
          <conf-name>Workshop on Large Language Models and Generative AI for Health at AAAI 2025</conf-name>
          <conf-date>March 4, 2025</conf-date>
          <conf-loc>Philadelphia, PA</conf-loc>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://openreview.net/pdf?id=ZcD35zKujO"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Cano</surname>
              <given-names>AH</given-names>
            </name>
            <name name-style="western">
              <surname>Romanou</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bonnet</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Matoba</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Salvi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Pagliardini</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Fang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Köpf</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mohtashami</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sallinen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sakhaeirad</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Swamy</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Krawczuk</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Bayazit</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Marmet</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Montariol</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hartley</surname>
              <given-names>M-A</given-names>
            </name>
            <name name-style="western">
              <surname>Jaggi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Bosselut</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>MEDITRON-70B: scaling medical pretraining for large language models</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on November 27, 2023</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2311.16079">https://arxiv.org/abs/2311.16079</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Touvron</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lavril</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Izacard</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Martinet</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Lachaux</surname>
              <given-names>M-A</given-names>
            </name>
            <name name-style="western">
              <surname>Lacroix</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Rozière</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Goyal</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Hambro</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Azhar</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Rodriguez</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Joulin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Grave</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lample</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>LLaMA: open and efficient foundation language models</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on February 27, 2023</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2302.13971">https://arxiv.org/abs/2302.13971</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bannett</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Gunturkun</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Pillai</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Herrmann</surname>
              <given-names>JE</given-names>
            </name>
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Huffman</surname>
              <given-names>LC</given-names>
            </name>
            <name name-style="western">
              <surname>Feldman</surname>
              <given-names>HM</given-names>
            </name>
          </person-group>
          <article-title>Applying large language models to assess quality of care: monitoring ADHD medication side effects</article-title>
          <source>Pediatrics</source>
          <year>2025</year>
          <month>01</month>
          <day>01</day>
          <volume>155</volume>
          <issue>1</issue>
          <fpage>e2024067223</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://publications.aap.org/pediatrics/article/155/1/e2024067223/200340/Applying-Large-Language-Models-to-Assess-Quality?autologincheck=redirected"/>
          </comment>
          <pub-id pub-id-type="doi">10.1542/peds.2024-067223</pub-id>
          <pub-id pub-id-type="medline">39701141</pub-id>
          <pub-id pub-id-type="pii">200340</pub-id>
          <pub-id pub-id-type="pmcid">PMC11978496</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Obermeyer</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Powers</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Vogeli</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mullainathan</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Dissecting racial bias in an algorithm used to manage the health of populations</article-title>
          <source>Science</source>
          <year>2019</year>
          <month>10</month>
          <day>25</day>
          <volume>366</volume>
          <issue>6464</issue>
          <fpage>447</fpage>
          <lpage>453</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://escholarship.org/uc/item/qt6h92v832"/>
          </comment>
          <pub-id pub-id-type="doi">10.1126/science.aax2342</pub-id>
          <pub-id pub-id-type="medline">31649194</pub-id>
          <pub-id pub-id-type="pii">366/6464/447</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Hardt</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Howell</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Chin</surname>
              <given-names>MH</given-names>
            </name>
          </person-group>
          <article-title>Ensuring fairness in machine learning to advance health equity</article-title>
          <source>Ann Intern Med</source>
          <year>2018</year>
          <month>12</month>
          <day>18</day>
          <volume>169</volume>
          <issue>12</issue>
          <fpage>866</fpage>
          <lpage>872</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://hdsi.uchicago.edu/wp-content/uploads/2021/07/Ensuring-Fairness-in-Machine-Learning-to-Advance-Health-Equity.pdf"/>
          </comment>
          <pub-id pub-id-type="doi">10.7326/m18-1990</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Du</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chuang</surname>
              <given-names>Y-W</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Guan</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lian</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Hong</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Bates</surname>
              <given-names>DW</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Testing and evaluation of generative large language models in electronic health record applications: a systematic review</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2026</year>
          <month>03</month>
          <day>01</day>
          <volume>33</volume>
          <issue>3</issue>
          <fpage>743</fpage>
          <lpage>753</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/jamia/article-lookup/doi/10.1093/jamia/ocaf233"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocaf233</pub-id>
          <pub-id pub-id-type="medline">41528313</pub-id>
          <pub-id pub-id-type="pii">8424252</pub-id>
          <pub-id pub-id-type="pmcid">PMC12981627</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Alnassar</surname>
              <given-names>SA</given-names>
            </name>
            <name name-style="western">
              <surname>Avison</surname>
              <given-names>KE</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>RS</given-names>
            </name>
            <name name-style="western">
              <surname>Raman</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Large language model applications for health information extraction in oncology: scoping review</article-title>
          <source>JMIR Cancer</source>
          <year>2025</year>
          <month>03</month>
          <day>28</day>
          <volume>11</volume>
          <fpage>e65984</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://cancer.jmir.org/2025//e65984/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/65984</pub-id>
          <pub-id pub-id-type="medline">40153782</pub-id>
          <pub-id pub-id-type="pii">v11i1e65984</pub-id>
          <pub-id pub-id-type="pmcid">PMC11970800</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wiest</surname>
              <given-names>IC</given-names>
            </name>
            <name name-style="western">
              <surname>Ferber</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>van Treeck</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Meyer</surname>
              <given-names>SK</given-names>
            </name>
            <name name-style="western">
              <surname>Juglan</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Carrero</surname>
              <given-names>ZI</given-names>
            </name>
            <name name-style="western">
              <surname>Paech</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kleesiek</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ebert</surname>
              <given-names>MP</given-names>
            </name>
            <name name-style="western">
              <surname>Truhn</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kather</surname>
              <given-names>JN</given-names>
            </name>
          </person-group>
          <article-title>Privacy-preserving large language models for structured medical information retrieval</article-title>
          <source>NPJ Digit Med</source>
          <year>2024</year>
          <month>09</month>
          <day>20</day>
          <volume>7</volume>
          <issue>1</issue>
          <fpage>257</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-024-01233-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-024-01233-2</pub-id>
          <pub-id pub-id-type="medline">39304709</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-024-01233-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC11415382</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jonnagaddala</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>ZSY</given-names>
            </name>
          </person-group>
          <article-title>Privacy preserving strategies for electronic health records in the era of large language models</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>01</month>
          <day>16</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>34</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01429-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01429-0</pub-id>
          <pub-id pub-id-type="medline">39820020</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01429-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC11739470</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Neveditsin</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Lingras</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Mago</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Clinical insights: a comprehensive review of language models in medicine</article-title>
          <source>PLOS Digit Health</source>
          <year>2025</year>
          <month>05</month>
          <day>8</day>
          <volume>4</volume>
          <issue>5</issue>
          <fpage>e0000800</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pdig.0000800"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pdig.0000800</pub-id>
          <pub-id pub-id-type="medline">40338967</pub-id>
          <pub-id pub-id-type="pii">PDIG-D-24-00307</pub-id>
          <pub-id pub-id-type="pmcid">PMC12061104</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Escorpizo</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Theotokatos</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Tucker</surname>
              <given-names>CA</given-names>
            </name>
          </person-group>
          <article-title>A scoping review on the use of machine learning in return-to-work studies: strengths and weaknesses</article-title>
          <source>J Occup Rehabil</source>
          <year>2024</year>
          <month>03</month>
          <volume>34</volume>
          <issue>1</issue>
          <fpage>71</fpage>
          <lpage>86</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://link.springer.com/article/10.1007/s10926-023-10127-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s10926-023-10127-1</pub-id>
          <pub-id pub-id-type="medline">37378718</pub-id>
          <pub-id pub-id-type="pii">10.1007/s10926-023-10127-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>IY</given-names>
            </name>
            <name name-style="western">
              <surname>Pierson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Rose</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Joshi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ferryman</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Ghassemi</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Ethical machine learning in healthcare</article-title>
          <source>Annu Rev Biomed Data Sci</source>
          <year>2021</year>
          <month>07</month>
          <day>20</day>
          <volume>4</volume>
          <issue>1</issue>
          <fpage>123</fpage>
          <lpage>144</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/34396058"/>
          </comment>
          <pub-id pub-id-type="doi">10.1146/annurev-biodatasci-092820-114757</pub-id>
          <pub-id pub-id-type="medline">34396058</pub-id>
          <pub-id pub-id-type="pmcid">PMC8362902</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>McHugh</surname>
              <given-names>ML</given-names>
            </name>
          </person-group>
          <article-title>Interrater reliability: the kappa statistic</article-title>
          <source>Biochem Med (Zagreb)</source>
          <year>2012</year>
          <volume>22</volume>
          <issue>3</issue>
          <fpage>276</fpage>
          <lpage>282</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.biochemia-medica.com/en/journal/22/3/10.11613/BM.2012.031"/>
          </comment>
          <pub-id pub-id-type="doi">10.11613/bm.2012.031</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kapoor</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Narayanan</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Leakage and the reproducibility crisis in machine-learning-based science</article-title>
          <source>Patterns (N Y)</source>
          <year>2023</year>
          <month>09</month>
          <day>08</day>
          <volume>4</volume>
          <issue>9</issue>
          <fpage>100804</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2666-3899(23)00159-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.patter.2023.100804</pub-id>
          <pub-id pub-id-type="medline">37720327</pub-id>
          <pub-id pub-id-type="pii">S2666-3899(23)00159-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC10499856</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kapoor</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cantrell</surname>
              <given-names>EM</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Pham</surname>
              <given-names>TH</given-names>
            </name>
            <name name-style="western">
              <surname>Bail</surname>
              <given-names>CA</given-names>
            </name>
            <name name-style="western">
              <surname>Gundersen</surname>
              <given-names>OE</given-names>
            </name>
            <name name-style="western">
              <surname>Hofman</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Hullman</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lones</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Malik</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Nanayakkara</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Poldrack</surname>
              <given-names>RA</given-names>
            </name>
            <name name-style="western">
              <surname>Raji</surname>
              <given-names>ID</given-names>
            </name>
            <name name-style="western">
              <surname>Roberts</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Salganik</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Serra-Garcia</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Stewart</surname>
              <given-names>BM</given-names>
            </name>
            <name name-style="western">
              <surname>Vandewiele</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Narayanan</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>REFORMS: consensus-based recommendations for machine-learning-based science</article-title>
          <source>Sci Adv</source>
          <year>2024</year>
          <month>05</month>
          <day>03</day>
          <volume>10</volume>
          <issue>18</issue>
          <fpage>eadk3452</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https:///www.science.org/doi/10.1126/sciadv.adk3452?url_ver=Z39.88-2003&amp;rfr_id=ori:rid:crossref.org&amp;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1126/sciadv.adk3452</pub-id>
          <pub-id pub-id-type="medline">38691601</pub-id>
          <pub-id pub-id-type="pmcid">PMC11092361</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gallifant</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Afshar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ameen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Aphinyanaphongs</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cacciamani</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Dligach</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Daneshjou</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Fernandes</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hansen</surname>
              <given-names>LH</given-names>
            </name>
            <name name-style="western">
              <surname>Landman</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lehmann</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>McCoy</surname>
              <given-names>LG</given-names>
            </name>
            <name name-style="western">
              <surname>Miller</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Moreno</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Munch</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Restrepo</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Savova</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Umeton</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Gichoya</surname>
              <given-names>JW</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Moons</surname>
              <given-names>KGM</given-names>
            </name>
            <name name-style="western">
              <surname>Celi</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Bitterman</surname>
              <given-names>DS</given-names>
            </name>
          </person-group>
          <article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title>
          <source>Nat Med</source>
          <year>2025</year>
          <month>01</month>
          <day>08</day>
          <volume>31</volume>
          <issue>1</issue>
          <fpage>60</fpage>
          <lpage>69</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id>
          <pub-id pub-id-type="medline">39779929</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-03425-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC12104976</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Moons</surname>
              <given-names>KGM</given-names>
            </name>
            <name name-style="western">
              <surname>Dhiman</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Riley</surname>
              <given-names>RD</given-names>
            </name>
            <name name-style="western">
              <surname>Beam</surname>
              <given-names>AL</given-names>
            </name>
            <name name-style="western">
              <surname>Van Calster</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Ghassemi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Reitsma</surname>
              <given-names>JB</given-names>
            </name>
            <name name-style="western">
              <surname>van Smeden</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Boulesteix</surname>
              <given-names>A-L</given-names>
            </name>
            <name name-style="western">
              <surname>Camaradou</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Celi</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Denaxas</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Denniston</surname>
              <given-names>AK</given-names>
            </name>
            <name name-style="western">
              <surname>Glocker</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Golub</surname>
              <given-names>RM</given-names>
            </name>
            <name name-style="western">
              <surname>Harvey</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Heinze</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Hoffman</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Kengne</surname>
              <given-names>AP</given-names>
            </name>
            <name name-style="western">
              <surname>Lam</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Loder</surname>
              <given-names>EW</given-names>
            </name>
            <name name-style="western">
              <surname>Maier-Hein</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Mateen</surname>
              <given-names>BA</given-names>
            </name>
            <name name-style="western">
              <surname>McCradden</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Oakden-Rayner</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ordish</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Parnell</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Rose</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Wynants</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Logullo</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods</article-title>
          <source>BMJ</source>
          <year>2024</year>
          <month>04</month>
          <day>16</day>
          <volume>385</volume>
          <fpage>e078378</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.bmj.com/lookup/pmidlookup?view=long&amp;pmid=38626948"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmj-2023-078378</pub-id>
          <pub-id pub-id-type="medline">38626948</pub-id>
          <pub-id pub-id-type="pmcid">PMC11019967</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Omar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sorin</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Agbareia</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Apakama</surname>
              <given-names>DU</given-names>
            </name>
            <name name-style="western">
              <surname>Soroush</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sakhuja</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Freeman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Horowitz</surname>
              <given-names>CR</given-names>
            </name>
            <name name-style="western">
              <surname>Richardson</surname>
              <given-names>LD</given-names>
            </name>
            <name name-style="western">
              <surname>Nadkarni</surname>
              <given-names>GN</given-names>
            </name>
            <name name-style="western">
              <surname>Klang</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Evaluating and addressing demographic disparities in medical large language models: a systematic review</article-title>
          <source>Int J Equity Health</source>
          <year>2025</year>
          <month>02</month>
          <day>26</day>
          <volume>24</volume>
          <issue>1</issue>
          <fpage>57</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://equityhealthj.biomedcentral.com/articles/10.1186/s12939-025-02419-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12939-025-02419-0</pub-id>
          <pub-id pub-id-type="medline">40011901</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12939-025-02419-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC11866893</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zack</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Lehman</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Suzgun</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rodriguez</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Celi</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Gichoya</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jurafsky</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Szolovits</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Bates</surname>
              <given-names>DW</given-names>
            </name>
            <name name-style="western">
              <surname>Abdulnour</surname>
              <given-names>RE</given-names>
            </name>
            <name name-style="western">
              <surname>Butte</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Alsentzer</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Assessing the potential of GPT-4 to perpetuate racial and gender biases in health care: a model evaluation study</article-title>
          <source>Lancet Digit Health</source>
          <year>2024</year>
          <month>01</month>
          <volume>6</volume>
          <issue>1</issue>
          <fpage>e12</fpage>
          <lpage>e22</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.sciencedirect.com/science/article/pii/S258975002300225X"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/s2589-7500(23)00225-x</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <collab>Canadian Institutes of Health Research (CIHR)</collab>
            <collab>Natural Sciences and Engineering Research Council of Canada (NSERC)</collab>
            <collab>Social Sciences and Humanities Research Council of Canada (SSHRC)</collab>
          </person-group>
          <source>Tri-Council Policy Statement: Ethical Conduct for Research Involving Humans, TCPS 2</source>
          <year>2022</year>
          <publisher-loc>Ottawa, Ontario, Canada</publisher-loc>
          <publisher-name>Secretariat on Responsible Conduct of Research</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Steinkamp</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Chambers</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Lalevic</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zafar</surname>
              <given-names>HM</given-names>
            </name>
            <name name-style="western">
              <surname>Cook</surname>
              <given-names>TS</given-names>
            </name>
          </person-group>
          <article-title>Automated organ-level classification of free-text pathology reports to support a radiology follow-up tracking engine</article-title>
          <source>Radiol Artif Intell</source>
          <year>2019</year>
          <month>09</month>
          <volume>1</volume>
          <issue>5</issue>
          <fpage>e180052</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/33937800"/>
          </comment>
          <pub-id pub-id-type="doi">10.1148/ryai.2019180052</pub-id>
          <pub-id pub-id-type="medline">33937800</pub-id>
          <pub-id pub-id-type="pmcid">PMC8017395</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>LY</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>XC</given-names>
            </name>
            <name name-style="western">
              <surname>Nejatian</surname>
              <given-names>NP</given-names>
            </name>
            <name name-style="western">
              <surname>Nasir-Moin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Abidin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Eaton</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Riina</surname>
              <given-names>HA</given-names>
            </name>
            <name name-style="western">
              <surname>Laufer</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Punjabi</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Miceli</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>NC</given-names>
            </name>
            <name name-style="western">
              <surname>Orillac</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Schnurman</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Livia</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Weiss</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Kurland</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Neifert</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Dastagirzada</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Kondziolka</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Cheung</surname>
              <given-names>ATM</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Cao</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Flores</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Costa</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Aphinyanaphongs</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Cho</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Oermann</surname>
              <given-names>EK</given-names>
            </name>
          </person-group>
          <article-title>Health system-scale language models are all-purpose prediction engines</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>07</month>
          <day>07</day>
          <volume>619</volume>
          <issue>7969</issue>
          <fpage>357</fpage>
          <lpage>362</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37286606"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06160-y</pub-id>
          <pub-id pub-id-type="medline">37286606</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06160-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC10338337</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Suchikova</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tsybuliak</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Teixeira da Silva</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Nazarovets</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>GAIDeT (Generative AI Delegation Taxonomy): a taxonomy for humans to delegate tasks to generative artificial intelligence in scientific research and publishing</article-title>
          <source>Account Res</source>
          <year>2026</year>
          <month>04</month>
          <volume>33</volume>
          <issue>3</issue>
          <fpage>2544331</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.tandfonline.com/doi/full/10.1080/08989621.2025.2544331"/>
          </comment>
          <pub-id pub-id-type="doi">10.1080/08989621.2025.2544331</pub-id>
          <pub-id pub-id-type="medline">40781729</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
