<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Res Protoc</journal-id><journal-id journal-id-type="publisher-id">ResProt</journal-id><journal-id journal-id-type="index">5</journal-id><journal-title>JMIR Research Protocols</journal-title><abbrev-journal-title>JMIR Res Protoc</abbrev-journal-title><issn pub-type="epub">1929-0748</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v15i1e96346</article-id><article-id pub-id-type="doi">10.2196/96346</article-id><article-categories><subj-group subj-group-type="heading"><subject>Protocol</subject></subj-group></article-categories><title-group><article-title>Extraction of Pain Severity and Functional Interference From Clinical Narratives Using Domain-Informed Large Language Models: Protocol for a Development and Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Xiaoyi</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wilson</surname><given-names>Christopher R</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Eyre</surname><given-names>Hannah</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Reed II</surname><given-names>David E</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hee Wai</surname><given-names>Travis Y</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kloehn</surname><given-names>Alexander</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Rosser</surname><given-names>Ethan W</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Luo</surname><given-names>Gang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Zeliadt</surname><given-names>Steven B</given-names></name><degrees>MPH, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Biomedical Informatics and Medical Education, University of Washington</institution><addr-line>Seattle</addr-line><addr-line>WA</addr-line><country>United States</country></aff><aff id="aff2"><institution>Seattle-Denver Center of Innovation for Veteran-Centered and Value-Driven Care, VA Puget Sound Health Care System</institution><addr-line>1660 S. Columbian Way</addr-line><addr-line>Seattle</addr-line><addr-line>WA</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Psychiatry and Behavioral Sciences, University of Washington</institution><addr-line>Seattle</addr-line><addr-line>WA</addr-line><country>United States</country></aff><aff id="aff4"><institution>Department of Health Systems and Population Health, School of Public Health, University of Washington</institution><addr-line>Seattle</addr-line><addr-line>WA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Fordham</surname><given-names>Bethany</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Ilodigwe</surname><given-names>Lucky</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Steven B Zeliadt, MPH, PhD, Seattle-Denver Center of Innovation for Veteran-Centered and Value-Driven Care, VA Puget Sound Health Care System, 1660 S. Columbian Way, Seattle, WA, 98108, United States, 1 2062774175; <email>steven.zeliadt@va.gov</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>7</day><month>10</month><year>2026</year></pub-date><volume>15</volume><elocation-id>e96346</elocation-id><history><date date-type="received"><day>01</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>21</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>22</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Xiaoyi Zhang, Christopher R Wilson, Hannah Eyre, David E Reed II, Travis Y Hee Wai, Alexander Kloehn,Ethan W Rosser, Gang Luo, Steven B Zeliadt. Originally published in JMIR Research Protocols (<ext-link ext-link-type="uri" xlink:href="https://www.researchprotocols.org">https://www.researchprotocols.org</ext-link>), 7.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Research Protocols, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.researchprotocols.org">https://www.researchprotocols.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.researchprotocols.org/2026/1/e96346"/><abstract><sec><title>Background</title><p>Chronic pain is a leading cause of disability and requires multidimensional assessment of pain intensity and functioning, yet electronic health records rarely capture these measures systematically. By contrast, surveys collecting patient-reported outcomes can assess pain over multiple dimensions but remain resource-intensive and difficult to scale for continuous population-level monitoring.</p></sec><sec><title>Objective</title><p>The objective of this study is to develop and validate a domain-informed natural language processing framework to derive pain severity and functional interference outcomes from unstructured clinical narratives. We aim to demonstrate that natural language processing&#x2013;derived outcomes can serve as a reliable, scalable surrogate for resource-intensive patient-reported surveys.</p></sec><sec sec-type="methods"><title>Methods</title><p>This study uses a retrospective cohort of 3725 Veterans with chronic musculoskeletal pain initiating complementary and integrative health therapies across 18 Veterans Health Administration Whole Health Flagship sites (2021&#x2010;2023). The dataset encompasses longitudinal patient-reported outcome surveys serving as the benchmark, linked with unstructured clinical narratives from the Veterans Health Administration electronic health record. Guided by established psychometric instruments and subject matter expert (SME) input, we developed a seed lexicon and annotation guidelines to identify language distinguishing 3 pain domains: pain severity, interference with enjoyment of life, and interference with general activities. Preliminary large language model (LLM) prompting was used to identify 600 candidate encounters (200 per domain) from 6747 notes across 260 patients for SME annotation, forming a ground truth validation sample. Two candidate LLMs will be evaluated on this sample; the best-performing LLM will generate a large library of span-level annotations to train a scalable, lightweight language model. The study uses a 3-stage validation process: (1) documentation completeness of pain interference in clinical narratives against SME-annotated references; (2) inference accuracy of the LLM-as-annotator and the fine-tuned lightweight model against SME annotations across note-level classification and span-level localization; and (3) concordance between the lightweight model&#x2019;s output and patient-reported pain, enjoyment, and general activity scores across a range of temporal windows.</p></sec><sec sec-type="results"><title>Results</title><p>As of July 2026, the cohort of 3725 Veterans has been identified and linked to clinical notes. The seed lexicon and annotation guidelines have been developed. Applying a developmental LLM to screen 6747 text notes in 6642 unique encounters over a 7-month period for 260 patients, at least 1 of the 3 pain domains was identified in 75% of notes and 99% of patients. SME validation at the encounter level is in progress. Final results from the subsequent knowledge distillation and validation stages are expected in the first half of 2027.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This protocol outlines a framework for identifying severe pain intensity and interference from clinical narratives, addressing a critical gap in health care system surveillance. To our knowledge, this is the first study to validate clinical text-based pain outcome extraction against patient-reported outcomes in a nationwide longitudinal cohort. If successful, this approach will enable health care systems to continuously monitor reports of pain-related functional interference and support more holistic, patient-centered pain management at scale.</p></sec><sec sec-type="registered-report"><title>International Registered Report Identifier (IRRID)</title><p>DERR1-10.2196/96346</p></sec></abstract><kwd-group><kwd>chronic pain</kwd><kwd>natural language processing</kwd><kwd>electronic health records</kwd><kwd>large language models</kwd><kwd>patient-reported outcomes</kwd><kwd>knowledge distillation</kwd><kwd>Veterans</kwd><kwd>pain assessment</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Chronic pain is a pervasive health issue that leads to poor patient outcomes and high health care costs. It affects one-fourth of adults in the United States, ranks as a leading cause of disability, and costs the nation over US $700 billion annually [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. The prevalence and severity of chronic pain among Veterans of the US military are disproportionately high compared with the general population [<xref ref-type="bibr" rid="ref4">4</xref>]. As such, chronic pain management is a high priority for health care providers and systems, including the Veterans Health Administration (VA) [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>As pain is a subjective, multidimensional experience, its assessment is complex but central to evaluating treatment efficacy and effectiveness. The IMMPACT (Initiative on Methods, Measurement, and Pain Assessment in Clinical Trials) consensus recommends assessment of multiple core outcome domains for chronic pain treatment trials, including pain (with its intensity, quality, and temporal aspects), physical functioning, and emotional functioning [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Assessing these domains, whether capturing the subjective experience of pain severity or outcomes concerning pain&#x2019;s interference with functioning (eg, general activity) and well-being (eg, enjoyment of life), relies heavily on patients&#x2019; self-reports.</p><p>While electronic health records (EHRs) are increasingly integrating self-reported pain data, with VA leading the shift from basic &#x201C;fifth vital sign&#x201D; scores [<xref ref-type="bibr" rid="ref8">8</xref>] to comprehensive measures [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>], systematic implementation of multidimensional patient-reported pain measures is rare [<xref ref-type="bibr" rid="ref11">11</xref>]. The commonly used Numerical Rating Scale (NRS) for pain intensity, a single-item numeric measurement, was intended for clinical screening and has limited utility for longitudinally tracking outcomes or capturing pain&#x2019;s broad interference in patient functioning and well-being [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>]. Although multidimensional instruments like the Brief Pain Inventory, PEG (Pain, Enjoyment, and General Activity) scale, and Patient-Reported Outcomes Measurement Information System (PROMIS) measures offer richer insights, they are typically administered in specific research settings or specialty clinics rather than in routine care [<xref ref-type="bibr" rid="ref13">13</xref>]. As pain is often discussed with providers as part of routine care, descriptions of pain interference, treatment plans, and associated progress may offer an alternative way to monitor the multidimensional aspects of pain management, supplementing targeted efforts to capture patient-reported pain outcomes.</p></sec><sec id="s1-2"><title>Prior Work</title><p>As part of a prior quality improvement initiative, VA conducted a nationwide survey across 18 VA medical centers participating in the Whole Health Flagship initiative, collecting patient-reported outcomes (PROs) from Veterans initiating complementary and integrative health (CIH) therapies [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. The survey used measures from the PEG scale, a validated and widely used instrument to assess pain severity and its interference with functioning and quality of life [<xref ref-type="bibr" rid="ref13">13</xref>]. Over the 2-year period spanning from March 2021 to March 2023, patient surveys were used to track longitudinal PROs from 3725 patients. Although providing valuable patient-reported data, surveys are resource-intensive, suffer from low response rates, and are difficult to scale for continuous population-level monitoring.</p><p>An alternative approach to obtaining outcomes equivalent to PROs at scale is through clinical notes, which contain narratives of patient reports and patient-provider discussions. Recent advances in machine learning and natural language processing (NLP) have generated techniques to extract pain and self-reported outcomes from narrative text data [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. Prior works have leveraged NLP methods to achieve objectives such as constructing an ontology of pain over diverse attributes [<xref ref-type="bibr" rid="ref19">19</xref>], identifying attributes of pain interference [<xref ref-type="bibr" rid="ref20">20</xref>], and using fine-tuned large language models (LLMs) to parse musculoskeletal pain features from unstructured clinical notes [<xref ref-type="bibr" rid="ref21">21</xref>]. Notably, recent studies using advanced Transformer-based deep learning models [<xref ref-type="bibr" rid="ref22">22</xref>] found that model-extracted features of pain experience were in agreement with domain experts [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>].</p><p>Despite the growing body of literature demonstrating the feasibility of using NLP to extract outcomes similar to PROs of chronic pain, several questions remain. First, documentation patterns for pain vary significantly across clinics and patient sociodemographic characteristics [<xref ref-type="bibr" rid="ref24">24</xref>]. How such variation manifests within the heterogeneous VA population and whether it introduces systematic bias into algorithmic models remains largely unknown. Second, while the broad notion of pain is frequently documented, documentation of concepts related to subjective aspects such as patient-reported functioning is reportedly more sparse compared with aspects like severity or site [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. A quantitative understanding of how completely clinical text data captures pain severity and functional interference domains remains underexamined. With these challenges in mind, we aim to develop a domain-informed NLP framework to operationalize these pain outcomes, leveraging linked longitudinal survey data as the benchmark to systematically quantify documentation completeness and validate the model&#x2019;s clinical utility.</p></sec><sec id="s1-3"><title>Study Objectives</title><p>The primary objective of this study is to develop a domain-informed NLP framework for operationalizing pain severity and functional interference outcomes directly from clinical text data that can be scaled at the population level. Leveraging our survey dataset of 3725 Veterans with longitudinal PEG measures and an encounter with the VA within a year of the survey as the benchmark, we pursue 3 specific aims:</p><list list-type="order"><list-item><p>Quantify how completely EHR clinical notes capture the 3 PEG domains of pain intensity and functional interference among patients with chronic pain engaging in pain management therapies.</p></list-item><list-item><p>Evaluate the accuracy of a best-performing LLM and a scalable, lightweight language model in extracting pain intensity and interference domains against subject matter expert (SME) annotations.</p></list-item><list-item><p>Assess the concordance between model-derived pain assessments from EHR encounters and proximal patient-reported PEG scores.</p></list-item></list><p>To our knowledge, this is the first large-scale study to validate clinical text-based pain outcome extraction against patient-reported measures in a longitudinal nationwide survey. Ultimately, we aim to establish the feasibility of using this NLP framework as a scalable, automated alternative to resource-intensive surveys, enabling health care systems to longitudinally monitor severe pain intensity and interference at the population level.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>This protocol outlines an ongoing study to develop and validate an NLP framework for identifying severe pain intensity and interference from clinical narratives across VA medical centers, using longitudinal PROs from a nationwide survey as the benchmark (<xref ref-type="fig" rid="figure1">Figure 1</xref>). At the time of writing, the project has finalized its methodological design and is currently in the model development phase. In the initial phase, we have completed the construct operationalization, translating PEG scale domains into a concrete seed lexicon derived from psychometrically validated instruments for use in developing annotation guidelines for identifying PEG-related concepts in clinical notes and rating each of the 3 PEG domains as either not severe (&#x003C;7/10) or severe (&#x2265;7/10).</p><p>Currently, we are in the process of adapting LLMs within VA&#x2019;s secure computing environment through prompt engineering, incorporating annotation guidelines to guide evidence extraction from clinical notes. Early LLM output has identified 600 candidate encounters (200 for each of the 3 pain domains) that are being verified by SMEs. This ground truth validation dataset will be used to assess the performance of 2 candidate LLMs. Although these resource-intensive LLMs are not currently scalable for reviewing EHR notes at the population level, this phase provides initial proof-of-concept validation. To enable scalable processing at the population level, we will adopt a knowledge distillation approach, using the best-performing LLM to generate a large library of span-level annotations that will be used to train a lightweight language model. This step will also allow transparency in reviewing which key language elements the LLM parses to determine severity levels within each of the 3 PEG domains. The lightweight model will then be assessed for performance relative to the best-performing LLM. Finally, we will compare the model-generated PEG domain ratings from EHR encounters to temporally adjacent PEG scores reported by patients to assess how closely EHR reports agree with PROs. Throughout these steps, we will conduct a 3-stage validation process to characterize the completeness of pain intensity and interference documentation in clinical narratives, evaluate the inference accuracy of the LLM-as-annotator and the lightweight model, and assess concordance between the lightweight model&#x2019;s output and PROs captured independently from the EHR. The completed TRIPOD+AI (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis and Artificial Intelligence) reporting checklist [<xref ref-type="bibr" rid="ref27">27</xref>] is provided in <xref ref-type="supplementary-material" rid="app4">Checklist 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Overview of the NLP framework for identifying severe pain intensity and interference from clinical narratives. (A) Formation of the study cohort and collection of PEG scores as the benchmark. (B) Development of the seed lexicon and annotation guidelines. (C) Model development and inference, including LLM prompt refinement on SME-annotated notes, LLM annotation of the remaining patient sample, and fine-tuning of a lightweight language model applied to the full cohort. (D) Validation pipeline for both the LLM and lightweight language models. CDW: Corporate Data Warehouse; CIH: complementary and integrative health; LLM: large language model; NLP: natural language processing; PEG: pain, enjoyment, and general activity; SME: subject matter expert; VA: Veterans Health Administration.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e96346_fig01.png"/></fig></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study constitutes a secondary analysis of PROs collected from the VA CIH Therapy Patient Experience Survey quality improvement evaluation [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. This evaluation was deemed a nonresearch activity under the Federal Policy for the Protection of Human Subjects and was initiated and executed as an internal quality improvement project for VHA operations in accordance with VA Program Guide 1200.21 [<xref ref-type="bibr" rid="ref28">28</xref>]. The present secondary analysis is conducted under an existing Memorandum of Understanding between the study team and the VA Office of Patient-Centered Care and Cultural Transformation, which governs data access for this work. No additional institutional review board review or waiver was required for this secondary analysis, as both the survey data from the initial evaluation project and the development of NLP evaluation tools to monitor pain interference in clinical narratives were deemed nonresearch.</p></sec><sec id="s2-3"><title>Data Governance and Privacy Safeguards</title><p>All analyses are conducted within VA&#x2019;s Health Insurance Portability and Accountability Act (HIPAA)&#x2013;compliant computing environment via the VA Informatics and Computing Infrastructure. All study data, including clinical notes, structured EHR fields, and survey responses, remain within VA firewalls at all times. Open-weight LLMs and the lightweight model are hosted on local VA graphics processing unit infrastructure, with no protected health information (PHI) transmitted to external API end points or third-party services; proprietary commercial LLMs accessed through external APIs are excluded from this study&#x2019;s pipeline. Access to the underlying Corporate Data Warehouse is role-based and limited to personnel credentialed by the VA to conduct internal quality improvement data analytics, with all access events logged through VA Informatics and Computing Infrastructure&#x2019;s standard auditing infrastructure.</p></sec><sec id="s2-4"><title>Study Setting</title><p>The Veterans Health Administration is the integrated health care system within the US Department of Veterans Affairs, providing care to enrolled Veterans through a national network of medical centers and outpatient sites. Enrollment is open to individuals with qualifying military service histories, although not all eligible Veterans enroll. Veterans receiving VA care carry a particularly heavy burden of chronic pain, with prevalence substantially higher than that of the general US adult population. This makes the cohort clinically important and represents a high-impact setting in which to develop scalable pain measurement tools. The Whole Health Flagship sites are 18 VA medical centers selected to lead implementation of VA&#x2019;s Whole Health System of Care, a patient-centered health care model emphasizing what matters to the Veteran rather than what is the matter with the Veteran. The 6 priority CIH therapies expanded under this initiative are acupuncture, chiropractic care, massage, Tai Chi, yoga, and meditation.</p><p>Data for this study were derived from a secondary analysis of longitudinal PROs collected by Office of Patient-Centered Care and Cultural Transformation&#x2019;s CIH Therapy Patient Experience Survey and corresponding EHRs from the VA Corporate Data Warehouse. The survey study protocol has been published [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. VA&#x2019;s Whole Health Flagship initiative, initially funded by the Comprehensive Addiction and Recovery Act of 2016, led to the expansion of CIH therapies across VA. The CIH Therapy Patient Experience Survey was conducted to assess the impact of this expansion on outcomes including pain severity and functional interference. Veterans invited to participate in the survey were aged 18 to 89 years, had been diagnosed with chronic musculoskeletal pain, and had newly initiated 1 of the 6 priority CIH therapies at any of the 18 VA medical centers participating in the Whole Health Flagship initiative.</p><p>Baseline surveys were sent on a weekly basis following identification of CIH therapy initiation between March 2021 and September 2022, with 6-month follow-up survey data collected through March 2023. Veterans included in this quality improvement evaluation were participants who had EHR-documented or self-reported use of a priority CIH therapy during the 6-month study period and who had completed self-reported pain assessment surveys at baseline and 6-month time points. This ensured that patients included in our study population had chronic pain and provided complete self-reported pain outcome data for analysis.</p></sec><sec id="s2-5"><title>Primary Pain Outcomes</title><p>The analysis focuses on replicating the individual domains of the PEG scale, a 3-item psychometric instrument consisting of 1 item for pain severity (&#x201C;P&#x201D;) and 2 items for pain interference with Enjoyment of life (&#x201C;E&#x201D;) and General activity (&#x201C;G&#x201D;) [<xref ref-type="bibr" rid="ref13">13</xref>] (<xref ref-type="fig" rid="figure2">Figure 2</xref>). For each domain, the goal is to determine whether there is information in the clinical narrative deemed to represent a patient&#x2019;s score of severe (&#x2265;7/10) or not severe (&#x003C;7/10). The threshold of 7 for severe was determined from prior literature, which observed that this threshold explained the highest proportion of variance in patient-reported interference, as introduced by Serlin et al [<xref ref-type="bibr" rid="ref31">31</xref>] and replicated in many different patient populations [<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref34">34</xref>]. Future efforts may address methods for combining the 3 domains into a summary score and identifying finer-level gradients of pain interference; because the 3 PEG domains are certain to be documented at different frequencies in clinical narratives relative to survey collection, where all domains are structurally asked at the same time, combining the individual domains extracted from clinical narratives is a substantially different analytic task that will depend on the intermediate findings of this analysis.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Pain, enjoyment, and general activity scale items administered in the complementary and integrative health Therapy Patient Experience Survey.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e96346_fig02.png"/></fig></sec><sec id="s2-6"><title>Developing an NLP Pipeline to Identify Pain Severity and Interference From Clinical Notes</title><sec id="s2-6-1"><title>Overview and Rationale</title><p>Processing clinical notes at the population level requires a scalable approach. LLMs demonstrate strong performance in understanding clinical text, but their high memory and inference time requirements make direct application to hundreds of thousands of notes impractical within current local infrastructure constraints. To address this issue, we adopt a practical design inspired by knowledge distillation, with the goal of developing a scalable, lightweight language model that can process large volumes of notes. To ensure the lightweight language model can accurately accomplish the nuanced task of evidence extraction, we will develop a comprehensive library of reference annotations using a combination of human review and expansion using an LLM applied to a sample larger than what would be feasible for efficient human review.</p></sec><sec id="s2-6-2"><title>Construct Operationalization</title><p>Pain interference is a multifaceted and abstract construct. Operationalizing this construct and extracting the related concrete information from clinical text present significant challenges. For instance, the PEG domain &#x201C;Enjoyment of life&#x201D; is a high-level concept that can be too abstract for direct extraction. To extract information relevant to pain intensity and interference as outlined in the PEG scale in a consistent and grounded manner, we developed an annotation guideline derived from multiple psychometric instruments, including item banks from the 41-item Patient-Reported Outcomes Measurement Information System-Pain Interference (PROMIS-PI) scale [<xref ref-type="bibr" rid="ref35">35</xref>], the 52-item West Haven-Yale Multidimensional Pain Inventory (WHYMPI) [<xref ref-type="bibr" rid="ref36">36</xref>], and the 10-item Oswestry Low Back Disability Questionnaire [<xref ref-type="bibr" rid="ref37">37</xref>]. These instruments were selected to provide specific and comprehensive coverage across the PEG domains. For example, PEG&#x2019;s &#x201C;Enjoyment of life&#x201D; (&#x201C;E&#x201D;) domain can be mapped to WHYMPI&#x2019;s concepts of &#x201C;satisfaction or enjoyment you get from participating in social and recreational activities.&#x201D; Similarly, for PEG&#x2019;s &#x201C;General activity&#x201D; (&#x201C;G&#x201D;) domain, the Oswestry Low Back Disability Questionnaire provides specific activity-based descriptions that align with documentation in clinical notes, such as &#x201C;I can only walk using crutches or a cane&#x201D; and &#x201C;Pain prevents me from standing for more than 10 mins.&#x201D; To capture PEG&#x2019;s pain intensity domain (&#x201C;P&#x201D;), we also included explicit severity descriptors (eg, &#x201C;severe pain,&#x201D; &#x201C;NRS 8/10&#x201D;) commonly used in clinical narratives. This guideline serves as the foundation for SME annotation and as the supervision signal for our language models. We curated concepts and verbatims from these instruments to form a comprehensive lexicon of 101 pain descriptors, as listed in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-6-3"><title>Validation Sample Development</title><p>We are developing a validation sample of 600 encounters, targeting 200 encounters for each of the 3 P, E, and G domains, with at least 75 severe (&#x2265;7) encounters within each domain. To efficiently generate this validation corpus, we sampled 260 patients from the survey population and identified all unique notes occurring between 1 month prior to completion of the baseline survey and 6 months following completion. To ensure relevance, all notes were required to have at least 1 instance of the word &#x201C;pain&#x201D; to be sampled. After this restriction, these patients had 6747 notes across 6642 unique encounters during this period. An early LLM applied to the 6747 text notes preliminarily identified notes with severe and not severe descriptions of interference across the 3 pain domains (<xref ref-type="table" rid="table1">Table 1</xref>).</p><p>Using eHost software [<xref ref-type="bibr" rid="ref38">38</xref>], annotation is being conducted by identifying sentence spans in clinical narratives that semantically map to concepts in the annotation guideline. Three SMEs with expertise in pain research are independently annotating clinical notes for a subset of patients sampled from the study population. Each SME reviews notes, identifies phrases representing PEG domains, and assigns severity to each domain using the annotation guide, grouped into &#x201C;Not severe (0&#x2010;6)&#x201D; and &#x201C;Severe (7-10)&#x201D; when appropriate descriptions are present in the text. Discrepancies are resolved through consensus adjudication among the SMEs.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Preliminary counts of pain, enjoyment, and general activity domains and severity levels from the developmental large language model applied to the validation sample.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">PEG<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> domain and severity level</td><td align="left" valign="bottom">Unique patients (N=260), n (%)</td><td align="left" valign="bottom">Total notes over 7-month period (N=6747)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Any of 3 PEG domains identified</td><td align="left" valign="top">257 (99)</td><td align="left" valign="top">5092 (75)</td></tr><tr><td align="left" valign="top">P (pain severity)&#x2014;any mention</td><td align="left" valign="top">257 (99)</td><td align="left" valign="top">4760 (71)</td></tr><tr><td align="left" valign="top">P (pain severity)&#x2014;severe</td><td align="left" valign="top">165 (63)</td><td align="left" valign="top">848 (13)</td></tr><tr><td align="left" valign="top">E (interference with enjoyment of life)&#x2014;any mention</td><td align="left" valign="top">187 (72)</td><td align="left" valign="top">960 (14)</td></tr><tr><td align="left" valign="top">E (interference with enjoyment of life)&#x2014;severe</td><td align="left" valign="top">50 (19)</td><td align="left" valign="top">88 (1)</td></tr><tr><td align="left" valign="top">G (interference with general activities)&#x2014;any mention</td><td align="left" valign="top">237 (91)</td><td align="left" valign="top">1938 (29)</td></tr><tr><td align="left" valign="top">G (interference with general activities)&#x2014;severe</td><td align="left" valign="top">98 (38)</td><td align="left" valign="top">287 (4)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>PEG: pain, enjoyment, and general activity.</p></fn><fn id="table1fn2"><p><sup>b</sup>Notes occurring on 6642 unique encounter dates.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-6-4"><title>Task Definition</title><p>Our pipeline involves 2 models, an LLM and a lightweight language model, each with distinct roles. After establishing proof of concept that the resource-intensive LLM can reliably identify elements of pain intensity and interference in clinical narratives, it will serve as an annotator (&#x201C;LLM-as-Annotator&#x201D;). Given the same information as provided to the SMEs, the LLM will receive free-text clinical narratives from our cohort. The expected output is extracted evidence mapped to the PEG domains, including relevant structured data when present within the note (eg, NRS scores) and specific sentence spans indicating pain intensity and interference, including markers of duration that represent severe interference such as &#x201C;most days.&#x201D; This will efficiently expand annotation capacity beyond what SMEs can achieve. An example input and annotation output is shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>. These annotations will then serve as the supervision signal for fine-tuning the second model, a lightweight language model&#x2014;such as a sentence transformer, a Bidirectional Encoder Representations from Transformers (BERT)&#x2013;based architecture, or a lightweight LLM (&#x003C;4B parameters)&#x2014;that performs both evidence extraction and severe pain classification for the entire cohort. We plan to use ClinicalBERT for this task.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Output of the LLM-as-annotator for development of the annotation training library. AUD: alcohol use disorder; BP: blood pressure; CC: chief complaint; EHR: electronic health record; HPI: history of present illness; LBP: low back pain; LLM: large language model; MDD: major depressive disorder; NRS: numeric rating scale; PDMP: prescription drug monitoring program; PEG: pain, enjoyment of life, and general activity; Pt: patient; PTSD: posttraumatic stress disorder; sx: symptoms; UDS: urine drug screen.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e96346_fig03.png"/></fig></sec><sec id="s2-6-5"><title>Data Preprocessing</title><p>We will retrieve clinical notes anchored to the same temporal windows surrounding each survey time point. The data representation will be concatenated with the system prompt to the LLM, along with subsequent instructions for formatting the output.</p></sec><sec id="s2-6-6"><title>Model Selection</title><p>We will leverage state-of-the-art open-weight (ie, models whose weights are publicly released, enabling local deployment) LLMs to perform this annotation task. Recent advances have demonstrated that LLMs pretrained on internet-scale corpora acquire extensive medical knowledge, with performance surpassing passing scores on professional medical examinations [<xref ref-type="bibr" rid="ref39">39</xref>-<xref ref-type="bibr" rid="ref42">42</xref>] and possessing robust medical reasoning capabilities [<xref ref-type="bibr" rid="ref43">43</xref>]. For our application scenario, we will compare 2 open-weight models that are available within the secure VA computing environment and demonstrate strong performance on a variety of benchmarks: Gemma 4 E4B [<xref ref-type="bibr" rid="ref44">44</xref>] and MedGemma 1.5 [<xref ref-type="bibr" rid="ref45">45</xref>]. Comparison between the 2 candidate LLMs will be based on their agreement with SME annotations, as described in the Three-Stage Validation section below. These models will be hosted on local machines, ensuring that all computation remains within VA&#x2019;s secure, HIPAA-compliant environment without transmitting PHI to external servers. For the downstream fine-tuned lightweight language model, we will evaluate lightweight architectures suitable for processing the entire cohort, as discussed in the <italic>Knowledge Distillation for Scalable Inference</italic> section below.</p></sec><sec id="s2-6-7"><title>Adapting LLMs to Generate Annotations</title><p>The LLMs will be adapted through prompt engineering. Core definitions from our annotation guidelines will be used to develop the prompt, with a structured JSON output schema aligned with the PEG domains. As a baseline, we will evaluate zero-shot performance using only prompt instructions. To expose the model to real-world documentation patterns, we will implement few-shot prompting with up to 5 representative examples selected from our development annotation set. These examples will cover diverse clinical scenarios and documentation styles, each pairing raw clinical notes with reference JSON outputs containing extracted evidence.</p><p>To mitigate hallucination risk, we will deploy a chain-of-thought (CoT) output schema requiring a sequential reasoning method [<xref ref-type="bibr" rid="ref46">46</xref>] where the model first identifies and quotes specific text spans indicating pain severity or functional interference, then maps this evidence to the corresponding PEG domains. By grounding outputs in cited evidence, the approach will enable systematic error analysis and iterative prompt refinement. This CoT reasoning is also important for complicated cases involving comorbidities. For example, social withdrawal documented in a mental health note may initially appear to stem from depression; however, if pain is documented elsewhere in the note as impacting the patient&#x2019;s mood, this functional limitation should also be considered as pain attributable. By requiring the model to first extract all relevant evidence before making attributions, the CoT schema ensures such contextual information is accounted for. The transparency granted by CoT reasoning also facilitates SME validation of LLM outputs, as human experts can directly audit the extracted evidence and reasoning process.</p></sec><sec id="s2-6-8"><title>Knowledge Distillation for Scalable Inference</title><p>To support scalable inference, we will adopt a knowledge distillation approach wherein LLM-generated annotations serve as the supervision signal for training a lightweight language model capable of processing the entire patient cohort. Our primary focus is ClinicalBERT [<xref ref-type="bibr" rid="ref47">47</xref>], a domain-specific BERT-based architecture [<xref ref-type="bibr" rid="ref48">48</xref>] pretrained on clinical and biomedical text, selected for its computational efficiency and its ability to be fully fine-tuned on local infrastructure to scale to population-level monitoring. Standard BERT-based models have a 512-token context limit, which may be insufficient for lengthy clinical notes. To address this, we will evaluate strategies including truncation to retain the most informative sections using rule-based sectionization found in medspaCy [<xref ref-type="bibr" rid="ref49">49</xref>], as well as long-context encoder architectures such as Clinical ModernBERT [<xref ref-type="bibr" rid="ref50">50</xref>], which supports up to 8192 tokens while maintaining the computational efficiency required for full-cohort processing.</p><p>Fine-tuning will be implemented via the Hugging Face Transformers library in an information extraction framework. Hyperparameters will be selected through k-fold cross-validation on the LLM-generated data, using <italic>F</italic><sub>1</sub>-score as the focus metric, with early stopping to mitigate overfitting. The highest-performing hyperparameter configuration, as measured by <italic>F</italic><sub>1</sub>-score, and the highest-performing overall configuration will be used to process all notes for the cohort, generating phrases containing PEG concepts and interference labels for all patients.</p></sec><sec id="s2-6-9"><title>Computational Feasibility and Optimization</title><p>Our local graphics processing unit infrastructure (4 NVIDIA L40S, 48 GB) supports inference of models up to 70B parameters at 16-bit floating-point precision. For further memory optimization, we may additionally implement 4-bit quantization [<xref ref-type="bibr" rid="ref51">51</xref>], which significantly reduces the memory footprint compared with full precision. This allows us to deploy all open-weight models on local machines, guaranteeing that all computation is contained entirely within the VA&#x2019;s secure, HIPAA-compliant environment without transmitting PHI to external servers.</p></sec><sec id="s2-6-10"><title>Three-Stage Validation for Our NLP Pipeline</title><sec id="s2-6-10-1"><title>Overview of 3-Stage Validation Process</title><p>To assess the reliability and clinical utility of our proposed approach, we will employ a 3-stage validation process. First, we will characterize the completeness of pain documentation in clinical narratives against SME-annotated references, quantifying how the 3 PEG domains are covered across notes, patients, and clinical settings. Second, we will evaluate the inference accuracy of the LLM-as-annotator, comparing 2 candidate LLMs (Gemma 4 E4B and MedGemma 1.5), and of the fine-tuned lightweight model against SME annotations across both note-level classification of pain domain and severity and span-level localization of pain-relevant text. Third, we will test the population-level utility of the fine-tuned lightweight model by assessing concordance between PROs captured contemporaneously and lightweight model classifications of proximal clinical narratives.</p><p>Stage 1 and Stage 2 validation activities will be conducted with the sample of 260 patients, along with their 600 sampled notes, who participated in the CIH Therapy Patient Experience Survey; Stage 3 will be conducted among the full sample of 3725 survey participants. The validation sample of 260 patients includes 6747 unique text notes that include the keyword &#x201C;pain&#x201D; over a 7-month period across 6642 unique encounters.</p></sec><sec id="s2-6-10-2"><title>Stage 1: Documentation of Pain Intensity and Interference in Clinical Narratives</title><p>The first validation measures the extent to which EHR documentation can be assessed for meaningful elements of pain intensity and pain interference, using the targeted set of 600 SME-annotated notes identified from the preliminary LLM prompt as the basis for this analysis. As noted in <xref ref-type="table" rid="table1">Table 1</xref>, the frequency of PEG domains may vary considerably, with the E domain reported less frequently. The goal of this validation stage is to confirm that the core elements defined in the annotation guidelines and used by the LLMs are reliably identified in clinical narratives. The analysis unit for this stage is the full encounter day, with SMEs reviewing all notes from an encounter date to ensure all contextual information is assessed. The most severe intensity or interference level on the day will be assessed by the SME. This activity will provide additional training for both refinement of the LLM prompts and span-level annotation for fine-tuning of the lightweight model, as SMEs will be asked to highlight specific text spans they used in their judgment to assess each of the PEG domains. We define the following metrics:</p><list list-type="bullet"><list-item><p>Interpretability: we report the proportion of notes in which pain was mentioned but SMEs could not determine domain or severity (&#x201C;indeterminate&#x201D;), and the proportion of spans requiring consensus adjudication.</p></list-item><list-item><p>Interrater reliability: prior to any adjudication, agreement among the 3 SMEs will be quantified using Fleiss &#x03BA; for the binary severity classification and for per-domain presence/absence. Because the sample is enriched for pain-relevant content by design, we note that these completeness metrics describe interpretable content within candidate notes rather than population-level documentation prevalence. Population-level documentation metrics will be assessed in subsequent validation stages using ClinicalBERT.</p></list-item></list><p>While preliminary examination of notes and our early LLM prompting have suggested there are extensive pain-related descriptions documented in clinical notes, this validation will shed light on potential gaps in EHR-documented clinical narratives. Potential limitations may arise due to limited provider documentation or insufficient descriptions by patients, leading to mentions of pain interference but uncertainty surrounding the extent of severity. To identify sources of bias prior to model deployment, we will examine how these metrics vary by clinical characteristics including age and sex.</p><p>In addition, we will record the proportion of notes in which a specific PEG domain is not mentioned as documentation gaps. Documentation gaps may arise from multiple sources, including limited symptom presence during the temporal window, provider documentation practices, and patient self-reporting patterns during clinical encounters. Our framework does not distinguish among these sources at the level of individual observations, as they collectively contribute to the observable evidence density from routine clinical documentation. Documentation gaps are characterized as observations of this evidence density rather than treated as missing data.</p></sec><sec id="s2-6-10-3"><title>Stage 2: Inference Accuracy of the LLM-as-Annotator and Lightweight Model Against SME Annotation</title><sec id="s2-6-10-3-1"><title>LLM-as-Annotator Evaluation</title><p>Conditional upon documentation completeness established in stage 1, this stage evaluates whether an LLM can serve as a reliable annotator, extending SME annotation capacity to a scale suitable for training a downstream lightweight model. Two candidate LLMs, Gemma 4 E4B [<xref ref-type="bibr" rid="ref44">44</xref>] and MedGemma 1.5 [<xref ref-type="bibr" rid="ref45">45</xref>], will be evaluated against the 600 SME-annotated notes across 2 complementary sub-analyses: note-level classification and span-level localization.</p></sec><sec id="s2-6-10-3-2"><title>Note-Level Classification of Domain and Severity</title><p>The purpose of this subanalysis is to demonstrate that an LLM can correctly identify descriptions of pain from text records and correctly classify its domain and severity. The LLM-as-annotator will be evaluated against the reference human annotations from the 600-encounter validation set to assess agreement with SME judgments (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Each LLM&#x2019;s performance will be calculated for each PEG domain and for binary level of severity if considered to contain elements suggestive of a level &#x2265;7, to validate how it performs against each of the individual domains and overall intensity. Precision, recall, and <italic>F</italic><sub>1</sub>-scores will be computed for each, with 95% CIs constructed by bootstrapping individuals and their corresponding notes, with replacement, to account for repeated measures within individuals (1000 replicates). An overall model-specific <italic>F</italic><sub>1</sub> measure will also be reported to compare the 2 candidate LLMs.</p></sec><sec id="s2-6-10-3-3"><title>Span-Level Localization</title><p>Conditional on the note/visit-level analysis identifying an LLM that can accurately identify the key PEG domains and levels of severity from patient notes, this subanalysis evaluates whether the LLM-as-annotator can identify the specific span of text and assign the correct pain domain and severity label. The goal is to demonstrate that using the LLM to develop a large dataset of span annotations maintains a high quality of accuracy. This large training dataset supports development of the lightweight language model, which can then scale to the entire patient cohort (<xref ref-type="fig" rid="figure3">Figure 3</xref>).</p><p>We first evaluate whether the LLM extracts the same pain-relevant text spans as our human SMEs. Given the generative nature of LLMs, outputs may not be verbatim copies of source text. We will employ 2 matching criteria:</p><list list-type="bullet"><list-item><p>Relaxed string match: A match is recorded if the extracted span shares tokens with the annotated span, relaxing exact boundary requirements.</p></list-item><list-item><p>Semantic equivalence: We will use an LLM-as-a-judge approach, prompting a separate LLM to determine whether extracted content preserves the same meaning as the annotation. A recent study has shown that LLM-as-a-judge provides a scalable way to identify accurate and safe LLM-generated clinical summaries [<xref ref-type="bibr" rid="ref52">52</xref>].</p></list-item></list><p>For both criteria, we will report precision, recall, and <italic>F</italic><sub>1</sub>-score. Precision measures the proportion of LLM-extracted spans that overlap with reference annotations. Any nonoverlapping extractions that do not exist in the source text constitute hallucinations. Recall measures the proportion of reference annotations successfully retrieved by the model. CIs for each performance metric will be constructed by bootstrapping individuals in the validation set, with replacement, to account for repeated observations within individuals (1000 replicates). For each of the 3 PEG domains, we will compare spans identified by the LLM to the SME annotations and report the full confusion matrix along with per-class precision, recall, and <italic>F</italic><sub>1</sub> with bootstrapped 95% CIs. We will separately report the same metrics for the binary severity label (severe vs not severe) for each of the 3 domains.</p><p>Once the fine-tuned lightweight model is completed, we will repeat both note-level classification and span-level localization subanalyses using the lightweight model against the same 600 SME-annotated encounters. This provides direct evaluation of the lightweight model&#x2019;s inference accuracy against the SME reference, alongside the LLM comparison.</p></sec></sec></sec></sec><sec id="s2-7"><title>Stage 3: Concordance Between the Scalable Lightweight Model and PROs</title><p>The lightweight model is designed for scalable, population-level deployment rather than individual-level diagnostic classification. Accordingly, the evaluation of concordance is framed around whether the model&#x2019;s output can serve as a reliable surrogate for PROs at population scale, complementing rather than replacing direct patient self-report. As such, we will assess the concordance between severe pain and interference identified by the lightweight model in clinical narratives and patient-reported PEG scores in the full 3725-patient cohort, with patient-reported scores serving as the benchmark (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). This analysis will be performed on the baseline data for each individual.</p><p>The primary discrimination metric is area under the receiver operating characteristic curve, computed individually for each PEG domain using the proportion of severe pain and interference notes within a survey-anchored window as the continuous ranking score against the patient&#x2019;s binary per-domain survey label (severe &#x2265;7/10 vs not severe &#x003C;7/10).</p><p>Patients without any pain-related clinical narrative within a given temporal window are excluded from that window&#x2019;s analysis, as no narrative observation exists. This exclusion is reflected in the reduced sample size at narrower windows (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Patients with pain-related narratives but without documentation of a specific PEG domain are retained as observations of low evidence density for that domain.</p><p>In addition to discrimination, we will assess calibration to evaluate how closely model-predicted rates of severe pain interference align with patient-reported rates at the aggregate level. Calibration will be reported through 3 complementary measures. Calibration plots will bin patients by decile of predicted probability and compare predicted rates against observed patient-reported rates within each bin. Brier scores will summarize overall probabilistic performance across the full cohort, with lower values indicating closer agreement between predicted and observed outcomes. Calibration-in-the-large will be computed as the ratio of predicted to observed severe pain interference prevalence, providing a single-number summary of systematic overprediction or underprediction.</p><p>Each analysis will be repeated for a range of temporal windows around the survey date (&#x00B1;7, &#x00B1;30, &#x00B1;90, and &#x00B1;180 d) to assess how the breadth of the aggregation window affects concordance. The 7-day window reflects the anchoring period of the PEG instrument, which asks patients to consider the past week. All point estimates will be reported with 95% CIs derived from a nonparametric bootstrap resampling patients as the independent unit over 1000 replicates.</p><p>Given the model&#x2019;s intended role as a population-scale complement to represent a similar concept as patient self-report of pain severity and interference, the following minimum performance criterion is calibrated to surrogate utility rather than individual-level diagnostic accuracy. The prespecified minimum performance criterion goal for Stage 3 is area under the receiver operating characteristic curve &#x2265;0.75 for each of the 3 domains, requiring discrimination meaningfully above chance while accommodating the attenuation expected when validating an EHR-derived surrogate against patients&#x2019; self-reports. We anticipate performance will deteriorate as encounters/clinical narratives become less temporally related to the survey date.</p><p>Performance will additionally be evaluated within prespecified subgroups of interest (age, sex, primary pain diagnosis, and clinic type) to determine whether the model systematically overestimates or underestimates severe pain interference for these subgroups. Subgroup differences will be assessed by comparing bootstrap-derived confidence intervals across subgroups.</p><p>Together, the 3 validation stages assess the full pipeline from documentation completeness to model inference accuracy and population-level utility. These results will determine the feasibility of monitoring severe pain and interference at population scale and whether our NLP framework can serve as a scalable surrogate for resource-intensive patient-reported surveys.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>This study is ongoing, with final results expected in 2027. As of July 2026, the study cohort of 3725 Veterans with chronic musculoskeletal pain who were new users of CIH therapies by survey administration date has been identified. The longitudinal PEG survey data serving as the benchmark have been fully aggregated. Demographic and clinical characteristics of the cohort have been previously published [<xref ref-type="bibr" rid="ref53">53</xref>].</p><p>The domain-specific seed lexicon has been developed, curating 101 pain descriptors from established psychometric instruments including PROMIS-PI, WHYMPI, and the Oswestry Low Back Disability Questionnaire. Annotation guidelines mapping these concepts to each of the 3 PEG domains have been developed and are being iteratively refined through engagement with SMEs. We have applied a developmental LLM to screen 6747 text notes among 6642 unique encounters over a 7-month period for 260 patients, with the preliminary LLM identifying at least 1 of the 3 PEG domains in 75% of notes and 99% of patients (<xref ref-type="table" rid="table1">Table 1</xref>). Validation at the note level, targeting 200 notes for each of the 3 PEG domains, is in progress as of July 2026. Upon completion of the validation-sample annotation, the study will proceed to the subsequent knowledge distillation and validation stages.</p></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Implications</title><p>Chronic pain is a complex health issue. Current screening tools embedded in structured EHR data cannot fully capture this complexity. Our framework extracts multidimensional pain information, including pain severity and interference with physical and emotional functioning, directly from clinical narratives at scale. If validated, this approach could offer a scalable alternative to resource-intensive surveys for monitoring severe pain and interference. To the best of our knowledge, this is the first study to investigate the feasibility of using EHR data, validated against large-scale PROs, for routine monitoring of severe pain and interference.</p></sec><sec id="s4-2"><title>Methodological Significance</title><p>This study adopts a domain-informed approach, grounding the LLM&#x2019;s inference in a seed lexicon derived from established psychometric instruments. This enables rigorous mapping of abstract pain-related PEG domains into concrete terms found in clinical documentation. To enable scalable processing, our approach employs knowledge distillation: an LLM first generates high-quality annotations on a subset of notes, which then serve as training data for a lightweight language model capable of processing the entire cohort. This design leverages LLM capabilities for nuanced evidence extraction while enabling scalable inference through a lightweight model. Our validation strategy extends beyond standard benchmarking against expert annotations. By first examining documentation completeness, we will assess whether disparities in clinical documentation practices, which have been identified in prior work [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>] and may introduce systemic bias, exist in our patient population. By validating against PROs, we will evaluate whether the framework&#x2019;s inferences align with how patients themselves report their pain and functional interference.</p></sec><sec id="s4-3"><title>Clinical and Operational Utility</title><p>Subject to successful validation, this framework could improve how pain is monitored in health care systems. Surveys collecting PROs often suffer from low response rates and nonresponse bias, as prior research found that nonresponders had higher pain-catastrophizing scores and more pain at enrollment [<xref ref-type="bibr" rid="ref12">12</xref>]. A validated NLP tool could serve as a passive approach to identify severe pain and functional interference, complementing active survey collection and reaching patients who are engaged in care but unreached by surveys. A validated severe pain phenotype would also enable longitudinal monitoring of treatment outcomes. While randomized controlled trials establish efficacy, they are limited in tracking long-term functional improvements at scale. By extracting pain and functional interference directly from EHR data, researchers could evaluate the efficacy of high-cost or invasive interventions such as spinal cord stimulation in reducing severe pain and interference in broader, real-world populations.</p></sec><sec id="s4-4"><title>Limitations</title><p>There are several limitations to this protocol. First, the LLMs&#x2019; inference relies on EHR documentation, which may be incomplete in recording the full picture of patients&#x2019; experience of pain. We address this limitation by quantifying documentation completeness in our validation. Second, pain fluctuates over time, and temporal gaps between clinical visits and survey dates may reduce concordance between EHR-derived and survey-reported outcomes. Third, significant heterogeneity in clinical documentation practices and reports of interference from pain will likely lead to underidentification of interference. Fourth, documentation biases in the EHR may propagate into model inference. If pain is systematically underdocumented for certain patient populations, the language model&#x2019;s inference for those groups may inherit this bias. We will conduct subgroup analyses by demographic and socioeconomic characteristics, comparing both documentation patterns (eg, the frequency and coverage of the PEG domains in annotations) and model inference accuracy metrics (eg, sensitivity, concordance with PROs) across groups to identify such disparities.</p><p>Fifth, patient representatives were not directly consulted during the development of the seed lexicon, annotation guidelines, or interpretation of model outputs in this secondary analysis, as the parent CIH Therapy Patient Experience Survey was conducted between 2021 and 2023 and has since concluded. This gap is partially mitigated by the fact that patient perspectives are embedded indirectly through the psychometric instruments used to construct the seed lexicon (PROMIS-PI, WHYMPI, Oswestry), each of which was developed with patient-engaged validation. Nonetheless, future deployment work would benefit from direct patient input on interpretability and face validity.</p><p>Finally, the direct comparison between aggregated NLP-derived domain evidence and PEG survey scores is an imperfect proxy for true concordance. Clinical documentation of pain is episodic and contextually driven, whereas the PEG reflects a patient&#x2019;s deliberate summary of their experience over time; simple aggregation of sparse span-level extractions may not fully bridge this representational gap. A natural extension of this work, contingent on demonstrating adequate NLP accuracy against SME annotations, would be to train a supervised model that learns to map heterogeneous, potentially sparse clinical documentation patterns onto PEG-equivalent scores, rather than relying on direct aggregation.</p></sec><sec id="s4-5"><title>Conclusion</title><p>This protocol outlines a framework for identifying severe pain intensity and interference from clinical narratives, addressing a gap in health care system surveillance. By grounding inference in established psychometric instruments and validating against PROs, we aim to develop an NLP-empowered tool that can reliably extract pain severity and functional interference from EHR data. If successful, this approach could provide a scalable complement to resource-intensive surveys, enabling longitudinal monitoring of severe pain interference at the population level.</p></sec></sec></body><back><ack><p>This work was conducted using resources and facilities of the VA Puget Sound Health Care System. The views expressed in this article are those of the authors and do not necessarily reflect the position or policy of the Department of Veterans Affairs or the United States Government.</p><p>The authors used generative AI tools (Claude Opus 4; Anthropic) for initial language editing and polishing of the manuscript text. All scientific content and final text were reviewed and verified by the authors, who take full responsibility for the publication.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the VA Health Systems Research Center of Innovation for Veteran-Centered and Value-Driven Care (COIN 13&#x2010;402) and the VA Office of Patient-Centered Care and Cultural Transformation (OPCC&#x0026;CT), Quality Enhancement Research Initiative (QUERI) award number PEC 13&#x2010;001. The funders provided support for investigator time, computing infrastructure, and, in the case of OPCC&#x0026;CT, access to the underlying CIH Therapy Patient Experience Survey data and linked electronic health records via the Memorandum of Understanding described in the Ethics Approval section. The funders had no role in the design of this secondary analysis, in the statistical analysis plan, in the interpretation of results, or in the decision to submit the manuscript for publication.</p></sec><sec><title>Data Availability</title><p>The data used in this study were obtained from the VA Corporate Data Warehouse and contain protected health information. In accordance with VA data governance policies, these data cannot be made publicly available. Requests for access to VA data may be directed to the VA Informatics and Computing Infrastructure [<xref ref-type="bibr" rid="ref54">54</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>All authors contributed to the intellectual development of this study. Individual contributions are as follows:</p><p>Conceptualization: XZ, CRW, HE, SBZ</p><p>Data curation: CRW, HE, AK</p><p>Funding acquisition: SBZ</p><p>Methodology: XZ, CRW, HE, DER, TYHW, AK, GL, SBZ</p><p>Project administration: AK, EWR</p><p>Software: CRW, HE</p><p>Supervision: GL, SBZ</p><p>Writing &#x2013; original draft: XZ</p><p>Writing &#x2013; review &#x0026; editing: CRW, HE, DER, TYHW, AK, EWR, GL, SBZ</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb2">CDW</term><def><p>Corporate Data Warehouse</p></def></def-item><def-item><term id="abb3">CIH</term><def><p>complementary and integrative health</p></def></def-item><def-item><term id="abb4">CoT</term><def><p>chain-of-thought</p></def></def-item><def-item><term id="abb5">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb6">HIPAA</term><def><p>Health Insurance Portability and Accountability Act</p></def></def-item><def-item><term id="abb7">IMMPACT</term><def><p>Initiative on Methods, Measurement, and Pain Assessment in Clinical Trials</p></def></def-item><def-item><term id="abb8">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb9">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb10">NRS</term><def><p>Numerical Rating Scale</p></def></def-item><def-item><term id="abb11">PEG</term><def><p>pain, enjoyment, and general activity</p></def></def-item><def-item><term id="abb12">PHI</term><def><p>protected health information</p></def></def-item><def-item><term id="abb13">PRO</term><def><p>patient-reported outcome</p></def></def-item><def-item><term id="abb14">PROMIS-PI</term><def><p>Patient-Reported Outcomes Measurement Information System-Pain Interference</p></def></def-item><def-item><term id="abb15">SME</term><def><p>subject matter expert</p></def></def-item><def-item><term id="abb16">TRIPOD+AI</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis and Artificial Intelligence</p></def></def-item><def-item><term id="abb17">VA</term><def><p>Veterans Health Administration</p></def></def-item><def-item><term id="abb18">WHYMPI</term><def><p>West Haven-Yale Multidimensional Pain Inventory</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lucas</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Sohi</surname><given-names>I</given-names> </name></person-group><article-title>Chronic pain and high-impact chronic pain in U.S. adults, 2023</article-title><source>NCHS Data Brief</source><year>2024</year><month>10</month><issue>518</issue><fpage>CS355235</fpage><pub-id pub-id-type="doi">10.15620/cdc/169630</pub-id><pub-id pub-id-type="medline">39751180</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guy</surname><given-names>GP</given-names>  <suffix>Jr</suffix></name><name name-style="western"><surname>Miller</surname><given-names>GF</given-names> </name><name name-style="western"><surname>Legha</surname><given-names>JK</given-names> </name><etal/></person-group><article-title>Economic costs of chronic pain-United States, 2021</article-title><source>Med Care</source><year>2025</year><month>09</month><day>1</day><volume>63</volume><issue>9</issue><fpage>679</fpage><lpage>685</lpage><pub-id pub-id-type="doi">10.1097/MLR.0000000000002181</pub-id><pub-id pub-id-type="medline">40730349</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mills</surname><given-names>SEE</given-names> </name><name name-style="western"><surname>Nicolson</surname><given-names>KP</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>BH</given-names> </name></person-group><article-title>Chronic pain: a review of its epidemiology and associated factors in population-based studies</article-title><source>Br J Anaesth</source><year>2019</year><month>08</month><volume>123</volume><issue>2</issue><fpage>e273</fpage><lpage>e283</lpage><pub-id pub-id-type="doi">10.1016/j.bja.2019.03.023</pub-id><pub-id pub-id-type="medline">31079836</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Taylor</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Kapos</surname><given-names>FP</given-names> </name><name name-style="western"><surname>Sharpe</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Kosinski</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Rhon</surname><given-names>DI</given-names> </name><name name-style="western"><surname>Goode</surname><given-names>AP</given-names> </name></person-group><article-title>Seventeen-year national pain prevalence trends among U.S. military veterans</article-title><source>J Pain</source><year>2024</year><month>05</month><volume>25</volume><issue>5</issue><fpage>104420</fpage><pub-id pub-id-type="doi">10.1016/j.jpain.2023.11.003</pub-id><pub-id pub-id-type="medline">37952861</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luther</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Finch</surname><given-names>DK</given-names> </name><name name-style="western"><surname>Bouayad</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Measuring pain care quality in the Veterans Health Administration primary care setting</article-title><source>Pain</source><year>2022</year><month>06</month><day>1</day><volume>163</volume><issue>6</issue><fpage>e715</fpage><lpage>e724</lpage><pub-id pub-id-type="doi">10.1097/j.pain.0000000000002477</pub-id><pub-id pub-id-type="medline">34724683</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dworkin</surname><given-names>RH</given-names> </name><name name-style="western"><surname>Turk</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Farrar</surname><given-names>JT</given-names> </name><etal/></person-group><article-title>Core outcome measures for chronic pain clinical trials: IMMPACT recommendations</article-title><source>Pain</source><year>2005</year><month>01</month><volume>113</volume><issue>1-2</issue><fpage>9</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.1016/j.pain.2004.09.012</pub-id><pub-id pub-id-type="medline">15621359</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wandner</surname><given-names>LD</given-names> </name><name name-style="western"><surname>Domenichiello</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Beierlein</surname><given-names>J</given-names> </name><etal/></person-group><article-title>NIH&#x2019;s Helping to End Addiction Long-term<sup>SM</sup> Initiative (NIH HEAL Initiative) Clinical Pain Management Common Data Element Program</article-title><source>J Pain</source><year>2022</year><month>03</month><volume>23</volume><issue>3</issue><fpage>370</fpage><lpage>378</lpage><pub-id pub-id-type="doi">10.1016/j.jpain.2021.08.005</pub-id><pub-id pub-id-type="medline">34508905</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="report"><article-title>Pain as the 5th Vital Sign Toolkit</article-title><year>2000</year><access-date>2025-11-12</access-date><publisher-name>Geriatrics and Extended Care Strategic Healthcare Group, National Pain Management Coordinating Committee, Veterans Health Administration</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.va.gov/painmanagement/docs/toolkit.pdf">https://www.va.gov/painmanagement/docs/toolkit.pdf</ext-link></comment></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mularski</surname><given-names>RA</given-names> </name><name name-style="western"><surname>White-Chu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Overbay</surname><given-names>D</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>L</given-names> </name><name name-style="western"><surname>Asch</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Ganzini</surname><given-names>L</given-names> </name></person-group><article-title>Measuring pain as the 5th vital sign does not improve quality of pain management</article-title><source>J Gen Intern Med</source><year>2006</year><month>06</month><volume>21</volume><issue>6</issue><fpage>607</fpage><lpage>612</lpage><pub-id pub-id-type="doi">10.1111/j.1525-1497.2006.00415.x</pub-id><pub-id pub-id-type="medline">16808744</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scher</surname><given-names>C</given-names> </name><name name-style="western"><surname>Meador</surname><given-names>L</given-names> </name><name name-style="western"><surname>Van Cleave</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Reid</surname><given-names>MC</given-names> </name></person-group><article-title>Moving beyond pain as the fifth vital sign and patient satisfaction scores to improve pain care in the 21st century</article-title><source>Pain Manag Nurs</source><year>2018</year><month>04</month><volume>19</volume><issue>2</issue><fpage>125</fpage><lpage>129</lpage><pub-id pub-id-type="doi">10.1016/j.pmn.2017.10.010</pub-id><pub-id pub-id-type="medline">29249620</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Owen-Smith</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mayhew</surname><given-names>M</given-names> </name><name name-style="western"><surname>Leo</surname><given-names>MC</given-names> </name><etal/></person-group><article-title>Automating collection of pain-related patient-reported outcomes to enhance clinical care and research</article-title><source>J Gen Intern Med</source><year>2018</year><month>05</month><volume>33</volume><issue>Suppl 1</issue><fpage>31</fpage><lpage>37</lpage><pub-id pub-id-type="doi">10.1007/s11606-018-4326-9</pub-id><pub-id pub-id-type="medline">29633139</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nugent</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Lovejoy</surname><given-names>TI</given-names> </name><name name-style="western"><surname>Shull</surname><given-names>S</given-names> </name><name name-style="western"><surname>Dobscha</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Morasco</surname><given-names>BJ</given-names> </name></person-group><article-title>Associations of pain numeric rating scale scores collected during usual care with research administered patient reported pain outcomes</article-title><source>Pain Med</source><year>2021</year><month>10</month><day>8</day><volume>22</volume><issue>10</issue><fpage>2235</fpage><lpage>2241</lpage><pub-id pub-id-type="doi">10.1093/pm/pnab110</pub-id><pub-id pub-id-type="medline">33749760</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krebs</surname><given-names>EE</given-names> </name><name name-style="western"><surname>Lorenz</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Bair</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>Development and initial validation of the PEG, a three-item scale assessing pain intensity and interference</article-title><source>J Gen Intern Med</source><year>2009</year><month>06</month><volume>24</volume><issue>6</issue><fpage>733</fpage><lpage>738</lpage><pub-id pub-id-type="doi">10.1007/s11606-009-0981-1</pub-id><pub-id pub-id-type="medline">19418100</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krebs</surname><given-names>EE</given-names> </name><name name-style="western"><surname>Carey</surname><given-names>TS</given-names> </name><name name-style="western"><surname>Weinberger</surname><given-names>M</given-names> </name></person-group><article-title>Accuracy of the pain numeric rating scale as a screening test in primary care</article-title><source>J Gen Intern Med</source><year>2007</year><month>10</month><volume>22</volume><issue>10</issue><fpage>1453</fpage><lpage>1458</lpage><pub-id pub-id-type="doi">10.1007/s11606-007-0321-2</pub-id><pub-id pub-id-type="medline">17668269</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Taylor</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Elwy</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Bokhour</surname><given-names>BG</given-names> </name><etal/></person-group><article-title>Measuring patient-reported use and outcomes from complementary and integrative health therapies: development of the Complementary and Integrative Health Therapy Patient Experience Survey</article-title><source>Glob Adv Integr Med Health</source><year>2024</year><volume>13</volume><fpage>27536130241241259</fpage><pub-id pub-id-type="doi">10.1177/27536130241241259</pub-id><pub-id pub-id-type="medline">38585239</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Calvert</surname><given-names>C</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Olson</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Complementary and integrative health therapies and pain: delivery through Veterans Affairs and community care</article-title><source>Glob Adv Integr Med Health</source><year>2025</year><volume>14</volume><fpage>27536130251358757</fpage><pub-id pub-id-type="doi">10.1177/27536130251358757</pub-id><pub-id pub-id-type="medline">40657238</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>SY</given-names> </name><etal/></person-group><article-title>Using artificial intelligence to improve pain assessment and pain management: a scoping review</article-title><source>J Am Med Inform Assoc</source><year>2023</year><month>02</month><day>16</day><volume>30</volume><issue>3</issue><fpage>570</fpage><lpage>587</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocac231</pub-id><pub-id pub-id-type="medline">36458955</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sim</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Horan</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Baker</surname><given-names>JN</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>IC</given-names> </name></person-group><article-title>Using natural language processing to analyze unstructured patient-reported outcomes data derived from electronic health records for cancer populations: a systematic review</article-title><source>Expert Rev Pharmacoecon Outcomes Res</source><year>2024</year><month>04</month><volume>24</volume><issue>4</issue><fpage>467</fpage><lpage>475</lpage><pub-id pub-id-type="doi">10.1080/14737167.2024.2322664</pub-id><pub-id pub-id-type="medline">38383308</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dave</surname><given-names>AD</given-names> </name><name name-style="western"><surname>Ruano</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kost</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name></person-group><article-title>Automated extraction of pain symptoms: a natural language approach using electronic health records</article-title><source>Pain Physician</source><year>2022</year><month>03</month><volume>25</volume><issue>2</issue><fpage>E245</fpage><lpage>E254</lpage><pub-id pub-id-type="medline">35322976</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Sim</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>JX</given-names> </name><etal/></person-group><article-title>Natural language processing and machine learning methods to characterize unstructured patient-reported outcomes: validation study</article-title><source>J Med Internet Res</source><year>2021</year><month>11</month><day>3</day><volume>23</volume><issue>11</issue><fpage>e26777</fpage><pub-id pub-id-type="doi">10.2196/26777</pub-id><pub-id pub-id-type="medline">34730546</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vaid</surname><given-names>A</given-names> </name><name name-style="western"><surname>Landi</surname><given-names>I</given-names> </name><name name-style="western"><surname>Nadkarni</surname><given-names>G</given-names> </name><name name-style="western"><surname>Nabeel</surname><given-names>I</given-names> </name></person-group><article-title>Using fine-tuned large language models to parse clinical notes in musculoskeletal pain disorders</article-title><source>Lancet Digit Health</source><year>2023</year><month>10</month><day>26</day><volume>5</volume><issue>12</issue><fpage>e855</fpage><lpage>e858</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00202-9</pub-id><pub-id pub-id-type="medline">39492289</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>N</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Guyon</surname><given-names>I</given-names> </name><name name-style="western"><surname>von Luxburg</surname><given-names>U</given-names> </name><name name-style="western"><surname>Bengio</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wallach</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fergus</surname><given-names>R</given-names> </name><name name-style="western"><surname>Vishwanathan</surname><given-names>SVN</given-names> </name><name name-style="western"><surname>Garnett</surname><given-names>R</given-names> </name></person-group><article-title>Attention is all you need</article-title><source>Advances in Neural Information Processing Systems 30 (NeurIPS 2017)</source><year>2017</year><publisher-name>Curran Associates Inc</publisher-name><fpage>5998</fpage><lpage>6008</lpage><pub-id pub-id-type="other">9781510860964</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amidei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Nieto</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kaltenbrunner</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ferreira De S&#x00E1;</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Serrat</surname><given-names>M</given-names> </name><name name-style="western"><surname>Albajes</surname><given-names>K</given-names> </name></person-group><article-title>Exploring the capacity of large language models to assess the chronic pain experience: algorithm development and validation</article-title><source>J Med Internet Res</source><year>2025</year><month>03</month><day>31</day><volume>27</volume><fpage>e65903</fpage><pub-id pub-id-type="doi">10.2196/65903</pub-id><pub-id pub-id-type="medline">40163858</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chaturvedi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Stewart</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ashworth</surname><given-names>M</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>A</given-names> </name></person-group><article-title>Distributions of recorded pain in mental health records: a natural language processing based study</article-title><source>BMJ Open</source><year>2024</year><month>04</month><day>19</day><volume>14</volume><issue>4</issue><fpage>e079923</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2023-079923</pub-id><pub-id pub-id-type="medline">38642997</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carlson</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Jeffery</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Characterizing chronic pain episodes in clinical text at two health care systems: comprehensive annotation and corpus analysis</article-title><source>JMIR Med Inform</source><year>2020</year><month>11</month><day>16</day><volume>8</volume><issue>11</issue><fpage>e18659</fpage><pub-id pub-id-type="doi">10.2196/18659</pub-id><pub-id pub-id-type="medline">33108311</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fodeh</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Finch</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bouayad</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Classifying clinical notes with pain assessment using machine learning</article-title><source>Med Biol Eng Comput</source><year>2018</year><month>07</month><volume>56</volume><issue>7</issue><fpage>1285</fpage><lpage>1292</lpage><pub-id pub-id-type="doi">10.1007/s11517-017-1772-1</pub-id><pub-id pub-id-type="medline">29280092</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Collins</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Moons</surname><given-names>KGM</given-names> </name><name name-style="western"><surname>Dhiman</surname><given-names>P</given-names> </name><etal/></person-group><article-title>TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods</article-title><source>BMJ</source><year>2024</year><month>04</month><day>16</day><volume>385</volume><fpage>e078378</fpage><pub-id pub-id-type="doi">10.1136/bmj-2023-078378</pub-id><pub-id pub-id-type="medline">38626948</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="report"><person-group person-group-type="author"><collab>Office of Research and Development</collab></person-group><article-title>VHA operations activities that may constitute research</article-title><year>2019</year><month>01</month><day>9</day><access-date>2026-09-12</access-date><publisher-name>Department of Veterans Affairs</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.research.va.gov/resources/policies/ProgramGuide-1200-21-VHA-Operations-Activities.pdf">https://www.research.va.gov/resources/policies/ProgramGuide-1200-21-VHA-Operations-Activities.pdf</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zeliadt</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Coggeshall</surname><given-names>S</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gelman</surname><given-names>H</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>SL</given-names> </name></person-group><article-title>The APPROACH trial: assessing pain, patient-reported outcomes, and complementary and integrative health</article-title><source>Clin Trials</source><year>2020</year><month>08</month><volume>17</volume><issue>4</issue><fpage>351</fpage><lpage>359</lpage><pub-id pub-id-type="doi">10.1177/1740774520928399</pub-id><pub-id pub-id-type="medline">32522024</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zeliadt</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Coggeshall</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gelman</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Assessing the relative effectiveness of combining self-care with practitioner-delivered complementary and integrative health therapies to improve pain in a pragmatic trial</article-title><source>Pain Med</source><year>2020</year><month>12</month><day>12</day><volume>21</volume><issue>Suppl 2</issue><fpage>S100</fpage><lpage>S109</lpage><pub-id pub-id-type="doi">10.1093/pm/pnaa349</pub-id><pub-id pub-id-type="medline">33313736</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Serlin</surname><given-names>RC</given-names> </name><name name-style="western"><surname>Mendoza</surname><given-names>TR</given-names> </name><name name-style="western"><surname>Nakamura</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>KR</given-names> </name><name name-style="western"><surname>Cleeland</surname><given-names>CS</given-names> </name></person-group><article-title>When is cancer pain mild, moderate or severe? Grading pain severity by its interference with function</article-title><source>Pain</source><year>1995</year><month>05</month><volume>61</volume><issue>2</issue><fpage>277</fpage><lpage>284</lpage><pub-id pub-id-type="doi">10.1016/0304-3959(94)00178-H</pub-id><pub-id pub-id-type="medline">7659438</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boonstra</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Stewart</surname><given-names>RE</given-names> </name><name name-style="western"><surname>K&#x00F6;ke</surname><given-names>AJA</given-names> </name><etal/></person-group><article-title>Cut-off points for mild, moderate, and severe pain on the Numeric Rating Scale for pain in patients with chronic musculoskeletal pain: variability and influence of sex and catastrophizing</article-title><source>Front Psychol</source><year>2016</year><volume>7</volume><fpage>1466</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2016.01466</pub-id><pub-id pub-id-type="medline">27746750</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hanley</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Masedo</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jensen</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Cardenas</surname><given-names>D</given-names> </name><name name-style="western"><surname>Turner</surname><given-names>JA</given-names> </name></person-group><article-title>Pain interference in persons with spinal cord injury: classification of mild, moderate, and severe pain</article-title><source>J Pain</source><year>2006</year><month>02</month><volume>7</volume><issue>2</issue><fpage>129</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1016/j.jpain.2005.09.011</pub-id><pub-id pub-id-type="medline">16459278</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jensen</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>DG</given-names> </name><name name-style="western"><surname>Ehde</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Robinsin</surname><given-names>LR</given-names> </name></person-group><article-title>Pain site and the effects of amputation pain: further clarification of the meaning of mild, moderate, and severe pain</article-title><source>Pain</source><year>2001</year><month>04</month><volume>91</volume><issue>3</issue><fpage>317</fpage><lpage>322</lpage><pub-id pub-id-type="doi">10.1016/S0304-3959(00)00459-0</pub-id><pub-id pub-id-type="medline">11275389</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amtmann</surname><given-names>D</given-names> </name><name name-style="western"><surname>Cook</surname><given-names>KF</given-names> </name><name name-style="western"><surname>Jensen</surname><given-names>MP</given-names> </name><etal/></person-group><article-title>Development of a PROMIS item bank to measure pain interference</article-title><source>Pain</source><year>2010</year><month>07</month><volume>150</volume><issue>1</issue><fpage>173</fpage><lpage>182</lpage><pub-id pub-id-type="doi">10.1016/j.pain.2010.04.025</pub-id><pub-id pub-id-type="medline">20554116</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kerns</surname><given-names>RD</given-names> </name><name name-style="western"><surname>Turk</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Rudy</surname><given-names>TE</given-names> </name></person-group><article-title>The West Haven-Yale Multidimensional Pain Inventory (WHYMPI)</article-title><source>Pain</source><year>1985</year><month>12</month><volume>23</volume><issue>4</issue><fpage>345</fpage><lpage>356</lpage><pub-id pub-id-type="doi">10.1016/0304-3959(85)90004-1</pub-id><pub-id pub-id-type="medline">4088697</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fairbank</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Couper</surname><given-names>J</given-names> </name><name name-style="western"><surname>Davies</surname><given-names>JB</given-names> </name><name name-style="western"><surname>O&#x2019;Brien</surname><given-names>JP</given-names> </name></person-group><article-title>The Oswestry low back pain disability questionnaire</article-title><source>Physiotherapy</source><year>1980</year><month>08</month><volume>66</volume><issue>8</issue><fpage>271</fpage><lpage>273</lpage><pub-id pub-id-type="medline">6450426</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>South</surname><given-names>B</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Leng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Forbush</surname><given-names>T</given-names> </name><name name-style="western"><surname>DuVall</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chapman</surname><given-names>W</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Cohen</surname><given-names>KB</given-names> </name><name name-style="western"><surname>Demner-Fushman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ananiadou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Webber</surname><given-names>B</given-names> </name><name name-style="western"><surname>Tsujii</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pestian</surname><given-names>J</given-names> </name></person-group><article-title>A prototype tool set to support machine-assisted annotation</article-title><source>BioNLP: Proceedings of the 2012 Workshop on Biomedical Natural Language Processing</source><year>2012</year><access-date>2026-09-12</access-date><publisher-name>Association for Computational Linguistics</publisher-name><fpage>130</fpage><lpage>139</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W12-2416/">https://aclanthology.org/W12-2416/</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="medline">37438534</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Nori</surname><given-names>H</given-names> </name><name name-style="western"><surname>King</surname><given-names>N</given-names> </name><name name-style="western"><surname>McKinney</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Carignan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Horvitz</surname><given-names>E</given-names> </name></person-group><article-title>Capabilities of GPT-4 on medical challenge problems</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 20, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.13375</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Oufattole</surname><given-names>N</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Szolovits</surname><given-names>P</given-names> </name></person-group><article-title>What disease does this patient have? A large-scale open domain question answering dataset from medical exams</article-title><source>Applied Sciences</source><year>2021</year><month>07</month><day>12</day><volume>11</volume><issue>14</issue><fpage>6421</fpage><pub-id pub-id-type="doi">10.3390/app11146421</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>AQ</given-names> </name><name name-style="western"><surname>Sablayrolles</surname><given-names>A</given-names> </name><name name-style="western"><surname>Roux</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Mixtral of experts</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 8, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2401.04088</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="web"><article-title>google/gemma-4-E4B</article-title><source>Hugging Face</source><year>2026</year><access-date>2026-05-28</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/google/gemma-4-E4B">https://huggingface.co/google/gemma-4-E4B</ext-link></comment></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sellergren</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mahvar</surname><given-names>F</given-names> </name><etal/></person-group><article-title>MedGemma 1.5 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 6, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2604.05081</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Schuurmans</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Chain-of-thought prompting elicits reasoning in large language models</article-title><source>Proc Int Conf Neural Inf Process Syst</source><year>2022</year><fpage>24824</fpage><lpage>24837</lpage><pub-id pub-id-type="doi">10.52202/068431-1800</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Alsentzer</surname><given-names>E</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boag</surname><given-names>W</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name></person-group><article-title>Publicly available clinical BERT embeddings</article-title><source>Proceedings of the 2nd Clinical Natural Language Processing Workshop</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>72</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.18653/v1/W19-1909</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Devlin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Toutanova</surname><given-names>K</given-names> </name></person-group><article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title><source>Proc Conf North Am Chapter Assoc Comput Linguist Hum Lang Technol</source><year>2019</year><fpage>4171</fpage><lpage>4186</lpage><pub-id pub-id-type="doi">10.18653/v1/N19-1423</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eyre</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chapman</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Peterson</surname><given-names>KS</given-names> </name><etal/></person-group><article-title>Launching into clinical space with medspaCy: a new clinical text processing toolkit in Python</article-title><source>AMIA Annu Symp Proc</source><year>2022</year><volume>2021</volume><fpage>438</fpage><lpage>447</lpage><pub-id pub-id-type="medline">35308962</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>JN</given-names> </name></person-group><article-title>Clinical ModernBERT: an efficient and long context encoder for biomedical text</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 4, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.03964</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dettmers</surname><given-names>T</given-names> </name><name name-style="western"><surname>Pagnoni</surname><given-names>A</given-names> </name><name name-style="western"><surname>Holtzman</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zettlemoyer</surname><given-names>L</given-names> </name></person-group><article-title>QLORA: efficient finetuning of quantized LLMs</article-title><source>Adv Neural Inf Process Syst</source><year>2023</year><fpage>10088</fpage><lpage>10115</lpage><pub-id pub-id-type="doi">10.52202/075280-0441</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Croxford</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>First</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Automating evaluation of AI text generation in healthcare with a large language model (LLM)-as-a-judge</article-title><source>medRxiv</source><year>2025</year><month>05</month><day>6</day><fpage>2025.04.22.25326219</fpage><pub-id pub-id-type="doi">10.1101/2025.04.22.25326219</pub-id><pub-id pub-id-type="medline">40313300</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zeliadt</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Coggeshall</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Bokhour</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Adding self-care complementary and integrative health therapies to care for chronic pain: the Assessing Pain, Patient Reported Outcomes and Complementary Health (APPROACH) study</article-title><source>Med Care</source><year>2026</year><month>05</month><day>1</day><volume>64</volume><issue>5</issue><fpage>283</fpage><lpage>292</lpage><pub-id pub-id-type="doi">10.1097/MLR.0000000000002295</pub-id><pub-id pub-id-type="medline">41771006</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="web"><article-title>VA Informatics and Computing Infrastructure (VINCI)</article-title><source>US Department of Veterans Affairs</source><access-date>2026-09-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.research.va.gov/programs/vinci/">https://www.research.va.gov/programs/vinci/</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Seed lexicon of 101 terms derived from validated pain assessment instruments and mapped to the 3 PEG domains (Pain severity, Enjoyment of life, General activity).</p><media xlink:href="resprot_v15i1e96346_app1.pdf" xlink:title="PDF File, 147 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Shell table&#x2014;encounter-level classification accuracy of candidate large language models against subject matter expert annotation for each pain severity, enjoyment of life, general activity domain and for binary severity level &#x2265;7.</p><media xlink:href="resprot_v15i1e96346_app2.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Shell table (partial)&#x2014;final validation reporting of Lightweight Model and survey responses.</p><media xlink:href="resprot_v15i1e96346_app3.docx" xlink:title="DOCX File, 15 KB"/></supplementary-material><supplementary-material id="app4"><label>Checklist 1</label><p>TRIPOD+AI checklist.</p><media xlink:href="resprot_v15i1e96346_app4.pdf" xlink:title="PDF File, 235 KB"/></supplementary-material></app-group></back></article>