<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Res Protoc</journal-id><journal-id journal-id-type="publisher-id">ResProt</journal-id><journal-id journal-id-type="index">5</journal-id><journal-title>JMIR Research Protocols</journal-title><abbrev-journal-title>JMIR Res Protoc</abbrev-journal-title><issn pub-type="epub">1929-0748</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v15i1e91675</article-id><article-id pub-id-type="doi">10.2196/91675</article-id><article-categories><subj-group subj-group-type="heading"><subject>Protocol</subject></subj-group></article-categories><title-group><article-title>Large Language Models in German Continuing Medical Education Assessments: Protocol for a Fully Crossed Experimental Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>&#x00D6;zmen</surname><given-names>Leyla</given-names></name><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Burisch</surname><given-names>Christian</given-names></name><degrees>Dr rer nat</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>G&#x00F6;dde</surname><given-names>Daniel</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Breuckmann</surname><given-names>Frank</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ehlers</surname><given-names>Jan</given-names></name><degrees>Prof Dr med vet</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sellmann</surname><given-names>Timur</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref></contrib></contrib-group><aff id="aff1"><institution>Faculty of Health, Witten/Herdecke University</institution><addr-line>Alfred-Herrhausen-Stra&#x00DF;e 50</addr-line><addr-line>Witten</addr-line><addr-line>North Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff2"><institution>Leibniz-Gymnasium Essen, District Government D&#x00FC;sseldorf</institution><addr-line>Essen</addr-line><addr-line>North-Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff3"><institution>Chair of Didactics and Educational Research in Healthcare, Witten/Herdecke University</institution><addr-line>Witten</addr-line><addr-line>North-Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff4"><institution>Department of Pathology and Molecular Pathology, HELIOS University Hospital Wuppertal, University Witten/Herdecke</institution><addr-line>Witten</addr-line><addr-line>North-Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff5"><institution>Department of Cardiology, Pneumology, Neurology and Intensive Care Medicine, Klinik Kitzinger Land</institution><addr-line>Kitzingen</addr-line><addr-line>Bayern</addr-line><country>Germany</country></aff><aff id="aff6"><institution>Department of Cardiology and Vascular Medicine, West German Heart and Vascular Center Essen, University Duisburg-Essen</institution><addr-line>Essen</addr-line><addr-line>North-Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff7"><institution>Department of Anaesthesiology I, Witten/Herdecke University</institution><addr-line>Witten</addr-line><addr-line>North-Rhine-Westphalia</addr-line><country>Germany</country></aff><aff id="aff8"><institution>Department of Anesthesiology and Intensive Care Medicine, Evangelisches Krankenhaus BETHESDA zu Duisburg</institution><addr-line>Duisburg</addr-line><country>Germany</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Arora</surname><given-names>Akshay</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Qi</surname><given-names>Wenhao</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Leyla &#x00D6;zmen, Faculty of Health, Witten/Herdecke University, Alfred-Herrhausen-Stra&#x00DF;e 50, Witten, North Rhine-Westphalia, 58455, Germany, 49 23029260; <email>Leyla.Oezmen@uni-wh.de</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>28</day><month>7</month><year>2026</year></pub-date><volume>15</volume><elocation-id>e91675</elocation-id><history><date date-type="received"><day>18</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>10</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>16</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Leyla &#x00D6;zmen, Christian Burisch, Daniel G&#x00F6;dde, Frank Breuckmann, Jan Ehlers, Timur Sellmann. Originally published in JMIR Research Protocols (<ext-link ext-link-type="uri" xlink:href="https://www.researchprotocols.org">https://www.researchprotocols.org</ext-link>), 28.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Research Protocols, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.researchprotocols.org">https://www.researchprotocols.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.researchprotocols.org/2026/1/e91675"/><abstract><sec><title>Background</title><p>Continuing medical education (CME) is a legal and ethical obligation for physicians in Germany. The rapid rise of large language models (LLMs) such as ChatGPT, Gemini, Claude, and Grok raises concerns about the integrity of CME assessments, as LLMs can already pass German CME tests.</p></sec><sec><title>Objective</title><p>This study aims to determine whether the choice of document format (searchable PDF, protected PDF, raster PDF, or vector PDF) and LLM influences the ability of LLMs to solve CME test questions at rates exceeding the passing threshold specified for each CME module (typically 70%).</p></sec><sec sec-type="methods"><title>Methods</title><p>In a fully crossed within-subjects repeated-measures design, 18 expired CME articles from 3 major German publishers across 6 specialties will be converted into 3 cheating-impeding PDF formats and processed alongside the original PDF files by 4 current LLMs (GPT-5, Claude Sonnet 4, Grok-4, and Gemini 3). This results in 16 model-format combinations. Each model will answer every article 3 times per file-format condition, with outcomes derived from aggregated run-level results. The primary outcome is the proportion of correctly answered questions; the secondary outcome is the pass/fail rate.</p></sec><sec sec-type="results"><title>Results</title><p>The study has been approved by the Witten/Herdecke University Ethics Committee (S-260/2025; dated August 10, 2025) and is preregistered at the Open Science Framework. The study is supported by internal departmental resources only, and no external funding was received. Because this protocol evaluates LLMs using expired CME materials, no human participants are being recruited. Data collection is planned to begin in June 2026 and is expected to last approximately 4 weeks. At the time of manuscript submission, no data have been collected or analyzed. Results are expected to be available after the completion of data collection and statistical analysis in 2026. The analyses will quantify performance differences across document formats; these findings may inform the feasibility of nonsearchable document formats as a temporary measure to reduce LLM-enabled cheating risks in CME contexts.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>By quantifying how document format constrains LLM performance, this study aims to evaluate simple technical safeguards that may reduce artificial intelligence&#x2013;assisted manipulation of CME tests and inform regulators and CME providers about how to balance assessment validity, accessibility, and responsible LLM integration into postgraduate medical education.</p></sec><sec><title>Trial Registration</title><p>OSF Registries osf.io/v96r5; https://osf.io/v96r5/overview</p></sec><sec sec-type="registered-report"><title>International Registered Report Identifier (IRRID)</title><p>DERR1-10.2196/91675</p></sec></abstract><kwd-group><kwd>continuing medical education</kwd><kwd>CME</kwd><kwd>large language models</kwd><kwd>LLMs</kwd><kwd>clinical competence assessment</kwd><kwd>educational measurement</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>medical education</kwd><kwd>Germany</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Large language models (LLMs) such as ChatGPT and Gemini represent transformative advances in natural language processing, demonstrating near-human proficiency across complex reasoning and knowledge-based tasks [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Recent studies have shown that ChatGPT achieves passing scores in medical board and licensing examinations, including the United States Medical Licensing Examination, often outperforming medical students [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>In Germany, continuing medical education (CME) is a mandatory component of the professional life of physicians. According to the Federal Medical Association (Bundes&#x00E4;rztekammer), licensed physicians must collect 250 CME points within 5 years, primarily through certified educational activities, including reading peer-reviewed CME articles and answering associated test questions [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>However, recent research revealed that nonmedical individuals using GPT-4 could successfully pass CME assessments [<xref ref-type="bibr" rid="ref5">5</xref>], highlighting vulnerabilities in CME evaluation systems. These findings raise ethical and methodological concerns, as CME credits might no longer reflect genuine physician learning or competence.</p><p>Building upon these observations, the present experiment evaluates whether the file format of CME materials impacts LLM performance. Specifically, it tests whether nonsearchable or graphically encoded PDFs (raster or vector) can serve as practical countermeasures to prevent LLM-assisted manipulation of CME assessments [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>The aim of this study is to determine whether the file format of CME materials (searchable, protected, raster, or vector PDF) modulates the ability of current-generation LLMs (GPT-5, Claude Sonnet 4, Grok-4, and Gemini 3) to correctly answer the associated multiple-choice questions at rates exceeding the 70% passing threshold typically required for CME credit. The null hypothesis states that technical measures have no impact on an LLM&#x2019;s ability to solve the CME tests. The alternative hypothesis posits that these measures impair this ability.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This study uses a fully crossed within-subjects repeated-measures design across 4 file-format conditions and 4 LLMs. Each CME article serves as its own control across file formats, yielding a within-item repeated-measures structure.</p><p>To account for the nondeterministic nature of LLMs, each model-format-article combination will be run 3 times, with primary and secondary outcomes being generated from recorded responses to CME test questions.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>The study follows the Declaration of Helsinki and Good Clinical Practice principles. Although no human participants are involved, the study adheres to these ethical principles to ensure transparency, scientific integrity, and responsible AI evaluation. Ethics approval was granted by the Witten/Herdecke University Ethics Committee (S-260/2025; dated October 8, 2025). The study is prospectively registered with the Open Science Framework (OSF; OSF.IO/V96R5).</p></sec><sec id="s2-3"><title>Study Material</title><p>The sampling frame will consist of expired CME modules made available by the 3 prespecified major German medical publishers included in this study (Deutscher &#x00C4;rzteverlag GmbH, Springer Medizin, and Georg Thieme Verlag). Eligible modules must contain (1) the complete CME article text, (2) the associated multiple-choice questions, and (3) the official answer key required for scoring. Only expired modules will be included to ensure that CME credit can no longer be claimed retrospectively. Modules that are still active, incomplete, duplicated, or technically unsuitable for standardized file-format conversion will be excluded.</p><p>To ensure structured coverage of major clinical domains, the eligible modules will be grouped according to 6 prespecified specialties: internal medicine, surgery, pediatrics, gynecology, neurology, and anesthesiology. Article selection will then be performed using a computer-generated randomization procedure from the eligible pool, with balanced allocation across specialties and publisher sources. This approach is intended to provide a heterogeneous multicenter sample from major German CME providers for a methodological comparison of file-format effects.</p><p>The aim of this sampling strategy is not to establish formal statistical representativeness of all German CME materials in Germany but rather to assemble a transparent, reproducible, and clinically diverse sample from major national CME providers under standardized experimental conditions.</p><p>All CME articles and the associated multiple-choice questions used in this study are protected by German copyright law (Urheberrechtsgesetz) and remain the intellectual property of the respective publishers (Deutscher &#x00C4;rzteverlag GmbH, Springer Medizin, and Georg Thieme Verlag). Prior to study initiation, all 3 publishers were formally contacted; were informed in writing about the nature, scope, and purpose of the study; and provided the requested expired CME modules together with explicit permission for their use in this research project. No specific conditions were imposed by the publishers. The corresponding correspondence is on file with the study team and is available upon reasonable request.</p><p>Only expired CME modules are included, that is, modules for which CME credit can no longer be claimed retrospectively. This ensures that the study cannot influence the active CME credit market, does not interfere with the publishers&#x2019; ongoing commercial interests, and excludes any possibility that LLM-generated answers could be misused to obtain valid CME credits. No original article text, question stems, distractors, or answer keys are reproduced in the manuscript, the supplementary materials, or the OSF repository. CME modules are identified only by anonymized internal codes together with their specialty and publication year range, and quantitative performance data are reported in aggregated form. All technically necessary reproductions generated for the file-format conversions (protected, raster, and vector PDFs) are used exclusively as input for the LLMs under evaluation. The CME materials are entered into the 4 LLMs (GPT-5, Claude Sonnet 4, Grok-4, and Gemini 3) via their official consumer web chat interfaces, in accordance with the CME providers&#x2019; terms of service for noncommercial research use. Where available, the interfaces are configured to disable use of submitted content for model training to prevent leakage of copyrighted material into future model versions.</p></sec><sec id="s2-4"><title>Technical Implementation of File Formats</title><p>Four document types will be generated for each CME article to enable within-article comparisons of file-format effects on LLM performance:</p><list list-type="order"><list-item><p>Searchable PDF: text-based PDFs in which characters are digitally encoded and machine-readable (PDF files as provided by the publishers)</p></list-item><list-item><p>Protected PDF: password-protected PDFs in which printing, modifying the content, and extracting text, images, and other elements are disabled (using 256-bit AES encryption)</p></list-item><list-item><p>Raster PDF: rasterized image-based PDFs. These consist of pixels arranged in a fixed grid, similar to digital photographs. Each pixel has a fixed color and position, and when zoomed in, the image becomes blurry due to the absence of additional information.</p></list-item><list-item><p>Vector PDF: PDFs that represent content using mathematical paths and B&#x00E9;zier curves. Vector graphics remain crisp and scalable, as shapes are dynamically recalculated rather than composed of static pixels.</p></list-item></list></sec><sec id="s2-5"><title>Protected PDF Generation and Verification</title><p>Protected PDFs will be generated by applying standardized security settings to the original publisher-provided files using 256-bit AES encryption. These settings will uniformly restrict content modification, printing, and extraction of text, images, and metadata across all documents. The applied protection parameters (including encryption level, permission flags, and software toolchain; eg, Adobe Acrobat [Adobe Inc] or equivalent) will be documented in detail in the appendix or OSF materials to ensure transparency and reproducibility.</p><p>The aim of the protected condition is not to alter the visual representation of the document but to assess whether access restrictions alone affect LLM performance despite the unchanged underlying content. Therefore, each protected PDF will undergo a verification step prior to model evaluation. This will include manual inspection to confirm that visual fidelity matches the source document and to verify that text extraction is effectively restricted or blocked under standard conditions. Successful protection will be defined as maintaining full human readability while preventing or substantially limiting direct programmatic access to the textual content.</p></sec><sec id="s2-6"><title>Raster PDF Generation and Verification</title><p>Raster PDFs will be generated using fixed rendering settings that are standardized across all articles, including prespecified image resolution and compression parameters, to ensure consistent image quality across the study. These settings will be documented in detail in the appendix or OSF materials to enable reproducibility.</p><p>The aim of the raster condition is not to create artificially unreadable files but to generate nonsearchable image-based PDFs that remain visually legible for human readers while limiting direct machine-readable text access. Therefore, each rasterized PDF will undergo a quality control check before model evaluation to confirm that page content remains readable and that rendering quality is not degraded to a level at which trivial optical character recognition (OCR) failure would be expected purely because of inadequate technical conversion settings.</p></sec><sec id="s2-7"><title>Vector PDF Generation and Verification</title><p>In the vector PDF condition, the goal is to preserve a &#x201C;vector-encoded&#x201D; visual representation while minimizing direct text extractability. To achieve this, all textual content will be converted into vector outlines (ie, text-to-path conversion) so that no selectable text layer remains. Vector PDFs will be generated using a standardized toolchain (eg, Adobe Acrobat Preflight and/or Ghostscript [Artifex Software] or Inkscape [Inkscape Project]). To verify the intended reduction in extractability, each generated vector PDF will undergo a quality control step using automated text-extraction checks (eg, pdftotext [Glyph &#x0026; Cog LLC] or equivalent); successful conversion will be defined as yielding no meaningful extracted text (eg, empty output or a negligible character count) while maintaining legibility comparable to the source document. These definitions are crucial, as LLMs differ in their ability to parse and extract text from graphical encodings. While raster formats may impede OCR and tokenization, vector formats retain structured information that might still be exploitable by advanced multimodal models [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref18">18</xref>].</p></sec><sec id="s2-8"><title>Participants and Models</title><p>The study does not involve human subjects as participants, only 1 &#x201C;operator&#x201D; (L&#x00D6;). Instead, the 4 &#x201C;participating&#x201D; LLMs (ChatGPT [version 5; OpenAI], Claude Sonnet [version 4; Anthropic PBC], Gemini [version 3; Google LLC], and Grok [version 4; xAI]) are treated as experimental agents. Each model will attempt to answer CME questions under 4 distinct format conditions. Responses will be collected and scored according to the official CME answer keys provided by the publishers.</p></sec><sec id="s2-9"><title>Reproducibility</title><p>To ensure reproducibility of the LLM experiments, we will provide the exact prompt templates used either verbatim in an appendix or via a permanent OSF appendix link. In brief, the prompt will include standardized task instructions, presentation of the current CME material in the assigned file-format condition, and instructions to answer the associated multiple-choice questions exclusively on the basis of the provided material. Core prompt components will be harmonized across models as far as technically feasible. Thus, the purpose of standardization is not to claim empirically proven prompt invariance across models but to minimize and transparently document prompt-related variation as consistently as possible across all study conditions.</p><p>Each model-format-article combination will be executed in 3 independent runs to account for stochastic variability in LLM outputs. These repeated runs will be used to derive the binary pass or fail outcome, as defined below. The primary analysis will use the mean accuracy across runs.</p></sec><sec id="s2-10"><title>Experimental Conditions</title><p>There are 16 model-format combinations, comprising a 4 (file format)&#x00D7;4 (LLM) fully crossed design. The four experimental conditions correspond to the 4 file-format types used in the within-article crossover design:</p><list list-type="order"><list-item><p>Condition 1 (searchable PDF condition): articles are presented as text-based PDFs accessible to all LLMs, as provided by the publishers.</p></list-item><list-item><p>Condition 2 (protected PDF condition): articles are presented as password-protected PDFs with disabled printing, extraction, and content modification.</p></list-item><list-item><p>Condition 3 (raster PDF condition): articles are provided as rasterized image-based PDFs, limiting direct text recognition.</p></list-item><list-item><p>Condition 4 (vector PDF condition): articles are rendered as vector PDFs, preserving structural outlines but reducing text extractability.</p></list-item></list></sec><sec id="s2-11"><title>Validity</title><p>The validity of the outcome measure depends critically on the LLMs&#x2019; ability to correctly parse and interpret the presented CME materials. Different PDF formats introduce systematic variation in this process: searchable PDFs provide a clean, tokenized text structure that can be processed directly, whereas rasterized PDFs require OCR, which is known to introduce transcription errors, reduce token accuracy, and increase noise in model inputs [<xref ref-type="bibr" rid="ref2">2</xref>]. To reduce the risk that observed performance losses merely reflect avoidable artifacts of excessively poor rasterization quality, raster PDFs will be generated using fixed standardized rendering settings and subjected to a prespecified quality control step before evaluation. Vector-encoded PDFs may further fragment text into glyphs or graphic paths, increasing parsing difficulty and occasionally exceeding tokenization limits in multimodal models [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. These format-dependent processing differences constitute a potential threat to construct validity, as performance may reflect a model&#x2019;s visual parsing capability rather than its medical reasoning competence. Therefore, evaluating file-format effects is necessary to isolate true diagnostic performance from artifacts introduced by differential input accessibility.</p></sec><sec id="s2-12"><title>Randomization</title><sec id="s2-12-1"><title>Overview</title><p>Randomization will occur on three levels: (1) the order in which articles are presented within each model-format sequence, (2) the allocation and order of file formats per article and model, and (3) the order of model evaluation. All randomization sequences will be computer-generated (R version 4.5.0; R Foundation for Statistical Computing; randomization blocks of equal size) and stored in a preregistered allocation file accessible only to the study statistician (see <xref ref-type="fig" rid="figure1">Figure 1</xref> for the CONSORT (Consolidated Standards of Reporting Trials)&#x2013;style flow diagram of article allocation, randomization, and model evaluation).</p><p>The 3 repeated runs for each model-format-article combination will be conducted as independent technical replicates in separate sessions with cleared context windows. They are intended to quantify stochastic output variability rather than to introduce an additional experimental factor. Run order will follow the randomized article-format-model sequence; no additional randomization of replicate order will be performed.</p><p>The study flow diagram in <xref ref-type="fig" rid="figure1">Figure 1</xref> illustrates the fully crossed within-subjects repeated-measures design with 18 CME articles, 4 PDF formats, and 4 LLMs, yielding 288 article-format-model combinations. Each combination will be executed in 3 independent runs, resulting in 864 total model runs.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>CONSORT (Consolidated Standards of Reporting Trials)&#x2013;style study flow diagram. CME: continuing medical education; LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="resprot_v15i1e91675_fig01.png"/></fig></sec><sec id="s2-12-2"><title>Randomization of CME Articles</title><p>Each of the 18 CME articles will be converted into all 3 cheating-impeding file formats (protected, raster, and vector), in addition to the original files provided by the publishers. For every article, a randomization list will specify the order in which the 4 formats are presented to each model. This yields a fully crossed within-article repeated-measures structure in which every model processes every article in each file-format condition in a triplicate manner (18 articles&#x00D7;4 formats&#x00D7;4 models&#x00D7;3 runs). Article order within each model-format sequence will be randomized using permuted blocks stratified by clinical specialty to maintain balance across internal medicine, surgery, pediatrics, gynecology, neurology, and anesthesiology.</p></sec><sec id="s2-12-3"><title>Randomization of File-Format Exposure</title><p>For each model and each CME article, the sequence of file formats (searchable, protected, raster, and vector) will be independently randomized using all 24 possible permutations. This ensures that no single format systematically benefits from warm-up or fatigue effects.</p></sec><sec id="s2-12-4"><title>Randomization of Model Evaluation Order</title><p>For each batch of articles, a unique model-order permutation (eg, GPT-5 to Gemini 3 to Claude Sonnet 4 to Grok-4) will be generated.</p></sec></sec><sec id="s2-13"><title>Prevention of Cross-Model Leakage</title><p>Models will never receive outputs, intermediate reasoning, or corrected answers from prior models. All responses are generated in isolated sessions with cleared context windows. No system prompts contain summaries or previous model outputs. In addition, within-model learning across file formats is minimized by enforcing strict session isolation: the same model never encounters more than 1 format of a given article within a single session, and no feedback on response correctness is provided between runs. Because all prompts start from an empty context window and contain only the current article in a single format, the risk of cumulative learning about individual CME items across formats is substantially reduced.</p></sec><sec id="s2-14"><title>Primary Outcome</title><p>The primary outcome is the proportion of correctly answered CME questions for each LLM across the 4 file-format conditions (searchable, protected, rasterized, and vector-encoded PDFs). Accuracy will be calculated as the percentage of items answered correctly relative to the official answer key provided by the publishers. For inferential analysis, the 3 repeated runs for each article-model-format combination will be treated as technical replicates and averaged before statistical testing. Thus, the primary inferential unit will be the article-level mean accuracy for each model and file-format condition, rather than the individual run. This approach accounts for stochastic run-to-run variability while avoiding artificial inflation of the effective sample size.</p></sec><sec id="s2-15"><title>Secondary Outcome</title><p>Pass/fail status will be determined using the passing threshold specified by the respective CME module/provider (typically 70%). The threshold for each module will be recorded from the module instructions. For the binary pass/fail outcome, each model-format-article combination will be classified as &#x201C;pass&#x201D; if at least 1 of the 3 runs meets the passing threshold; otherwise, it will be classified as &#x201C;fail.&#x201D; This classification takes into account the models&#x2019; fundamental ability to pass a CME test despite the measures in effect. For inferential analysis, pass or fail status will therefore be evaluated at the article-model-format level rather than at the individual-run level.</p></sec><sec id="s2-16"><title>Statistical Analysis</title><sec id="s2-16-1"><title>Sample Size and Power Calculation</title><p>The primary aim of this study is to quantify the effect of document format on LLM accuracy (searchable, protected, raster, and vector PDFs) within a fully crossed 4 (format)&#x00D7;4 (LLM) repeated-measures design, in which each article serves as its own control across formats. The study includes 18 CME articles, each evaluated under all 16 model-format combinations, with 3 repeated technical runs per article-model-format combination.</p><p>The repeated runs will not be treated as independent inferential observations but will be aggregated before hypothesis testing. For the primary outcome, the inferential unit is the CME article within each LLM and file-format condition. Therefore, each predefined comparison of an intervention format against the searchable PDF condition will be based on 18 paired article-level observations per LLM. The sample size calculation was performed for the hypothesis that the use of LLMs would improve CME test results from the guessing probability (20% correct answers, SD &#x03C3;=40%) to the passing level (70% correct answers) with a confidence level of 1&#x2013;&#x03B1;=0.95 (ie, &#x03B1;=.05) and a high statistical power of 1<italic>&#x2013;</italic>&#x03B2;&#x003E;0.95.</p><p>Because CME courses can be completed multiple times without changes to their content or assessment questions, we used a paired design in which randomly selected tests were administered identically across all 16 experimental conditions. To provide a more conservative and clinically diverse sample, we included 18 CME articles. We reviewed the sample size for the binary pass or fail approach of the secondary outcome, again with a confidence level of 1<italic>&#x2013;</italic>&#x03B1;=0.95 (ie, &#x03B1;=.05) and a high statistical power of 1<italic>&#x2013;</italic>&#x03B2;&#x003E;0.95.</p><p>A passing probability of 100% was observed in a previous study when an LLM was provided with the complete CME material, which will be the case in this study. Because newer LLM generations are expected to perform at least as well as the models evaluated in the previous study, we conservatively assumed an 80% passing probability under full access to the CME material. If the cheating-impeding measures are effective, performance is expected to drop to a binomial probability of approximately 0.086% for achieving &#x2265;70% correct answers by chance. To remain conservative, we assumed a 20% passing probability under effective cheating-impeding conditions for the sample size calculation.</p><p>Under these very conservative settings, we calculated a sample size of 7 CME modules. We again chose a conservative approach and included all 18 CME modules from the previous study.</p><p>The final sample size is considered sufficient for detecting substantial, practically meaningful file-format effects while acknowledging that the study is not designed to detect small higher-order interaction effects.</p></sec><sec id="s2-16-2"><title>Statistical Testing</title><p>Descriptive statistics will summarize accuracy and pass/fail rates across LLMs and file-format conditions. Individual run-level results will be retained for descriptive reporting of stochastic variability, but they will not be treated as independent inferential observations.</p><p>For the primary outcome, the proportion of correctly answered CME questions will first be calculated for each run. The 3 runs for each article-model-format combination will then be averaged. For each LLM and each file-format condition, the inferential dataset will therefore consist of 18 article-level mean accuracy values.</p><p>The primary comparisons will be performed separately for each LLM. Within each LLM, the 3 cheating-impeding file formats&#x2014;protected PDF, raster PDF, and vector PDF&#x2014;will each be compared pairwise with the searchable PDF condition. Because the same CME articles are evaluated under all file-format conditions, comparisons will be performed as paired within-article analyses. For each article, the difference in mean accuracy between the intervention format and the searchable PDF condition will be calculated. Depending on the distribution of these paired differences, paired <italic>t</italic> tests or Wilcoxon signed-rank tests will be used. Given the small sample size of 18 articles per LLM, Wilcoxon signed-rank tests will be preferred if normality of the paired differences is not supported.</p><p>For the secondary binary pass/fail outcome, pass/fail status will be determined for each article-model-format combination according to the prespecified module-specific passing threshold. For each LLM and file-format condition, the number and percentage of passed CME modules among the 18 articles will be reported. Each intervention format will then be compared with the searchable PDF condition within the same LLM using paired binary analyses. Because the same CME articles are evaluated under both file-format conditions, McNemar-type tests will be used. Given the small number of articles, exact binomial tests based on discordant pairs will be preferred.</p><p>Multiplicity adjustment will be applied to the 3 predefined comparisons against the searchable PDF condition within each LLM. Holm adjustment will be used as the primary multiplicity correction. Adjusted <italic>P</italic> values &#x003C;.05 will be considered statistically significant. All analyses will be performed in R.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>The study has been approved by the Witten/Herdecke University Ethics Committee (S-260/2025; dated August 10, 2025) and is preregistered at the Open Science Framework. The study is supported by internal departmental resources only, and no external funding was received. Because this protocol evaluates LLMs using expired CME materials, no human participants are being recruited. Data collection is planned to begin in July 2026 and is expected to last approximately 4 weeks. At the time of manuscript submission, no data have been collected or analyzed. Results are expected to be available after the completion of data collection and statistical analysis in 2026. The analyses will quantify performance differences across document formats (searchable, protected, rasterized, and vector-encoded PDFs); these findings may inform the feasibility of nonsearchable document formats as a temporary measure to reduce LLM-enabled cheating risks in CME contexts.</p></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>This protocol describes a study designed to test the hypothesis that searchable PDF formats will enable the highest LLM performance on German CME assessments, whereas protected, rasterized, and vector-encoded PDFs will reduce performance by limiting direct machine readability. We further expect that the magnitude of this format effect may vary across model families, depending on their multimodal input-processing capabilities and robustness to OCR-related or layout-related degradation.</p><p>If confirmed, these findings would suggest that document format is not a neutral technical property but a potentially relevant determinant of AI solvability in postgraduate medical assessment environments. In this sense, the study is expected to contribute not only to the evaluation of LLM performance in medical education but also to the broader question of whether simple file-format modifications could function as low-threshold safeguards against AI-assisted misuse in CME settings.</p></sec><sec id="s4-2"><title>Limitations</title><p>This study has several methodological limitations that must be acknowledged. First, LLMs are inherently nondeterministic. Even with fixed prompts and identical input materials, models may produce slightly different outputs across runs due to stochastic sampling, internal state variation, or backend optimization routines.</p><p>The study does not include a human control group, preventing direct comparison of how physicians perform with protected, rasterized, or vector-based PDFs compared to LLMs. The experimental setup uses standardized prompts and does not examine the impact of advanced prompt engineering or multistep attacks (eg, external OCR followed by LLM processing), which could potentially circumvent format-based restrictions in real-world cheating scenarios. Although the conflict between AI protection and accessibility for visually impaired users is discussed, it is not empirically investigated.</p><p>Finally, modern LLMs are subject to frequent backend updates and performance drift over time, even when model names remain unchanged. As a result, the behavior of the evaluated models at the time of data collection (planned for June 2026) may differ from their performance at the time of publication or future replication attempts. We will therefore document all model version identifiers and run time stamps in detail, but residual version drift remains an inherent limitation of AI evaluation studies.</p></sec><sec id="s4-3"><title>Comparison With Prior Work</title><p>If protected, rasterized, or vector-encoded CME materials can be shown to reduce LLM performance, such formats may constitute a simple technical safeguard for protecting the integrity of CME assessments by limiting machine readability. However, implementing deliberately nonsearchable formats introduces a significant accessibility conflict. Under the German Accessible Information Technology Ordinance (Barrierefreie-Informationstechnik-Verordnung) and other technical guidelines [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>], educational materials provided by professional bodies should, where feasible, remain accessible to users with visual impairments, age-related reading limitations, or reliance on assistive technologies such as screen readers or text-to-speech systems. Nonsearchable or access-restricted PDF formats may impair these functions by restricting text extraction, semantic navigation, adjustable contrast, and integrated search tools. Consequently, any attempt to constrain LLM capabilities through file-format manipulation must be balanced against the legal and ethical obligation to ensure accessibility and equal participation for human learners. Future studies should explore hybrid countermeasures that combine technical barriers with authentication protocols (eg, CAPTCHA-style test verification). Beyond regulatory implications, this research contributes to broader discussions on AI governance and digital literacy in postgraduate medical education [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref27">27</xref>].</p></sec><sec id="s4-4"><title>Future Directions</title><p>Future studies should determine whether any observed file-format effects remain stable across newer model generations, other medical specialties, and additional assessment settings beyond CME article-based testing. It will also be important to investigate whether format-based restrictions can be circumvented through combined workflows, such as external OCR, multimodal preprocessing, or prompt-engineered extraction strategies.</p><p>In addition, future research should evaluate human-centered alternatives that preserve accessibility while maintaining assessment integrity, including secure authentication procedures, adaptive item delivery, hybrid anticheating measures, and accessibility-preserving technical safeguards. Comparative studies involving human participants may further clarify whether format manipulations disproportionately affect legitimate learners, particularly those who rely on assistive technologies.</p></sec><sec id="s4-5"><title>Dissemination Plan</title><p>The findings of this study will be disseminated through submission to a peer-reviewed journal and presentation at scientific meetings in the fields of medical education, digital health, and health professions assessment. In addition, the results will be communicated to relevant CME stakeholders, including publishers, professional bodies, and regulatory audiences, to inform ongoing discussions about assessment integrity, accessibility, and responsible AI governance in postgraduate medical education. Where appropriate, study materials relevant to reproducibility will be made available via the OSF in accordance with journal and copyright requirements.</p></sec><sec id="s4-6"><title>Conclusions</title><p>Rather than merely describing a technical comparison of PDF types, this protocol addresses a timely governance question in postgraduate medical education: whether assessment materials can be made less vulnerable to AI-assisted misuse without creating unacceptable barriers for human users. By examining how searchable, protected, rasterized, and vector-encoded formats affect LLM performance, the study is expected to generate evidence relevant to the design of CME assessments at the intersection of validity, accessibility, and responsible AI integration. The findings may help inform future policy and technical decisions by CME providers, publishers, and regulators.</p></sec></sec></body><back><ack><p>This work forms part of the doctoral thesis of LO at the Faculty of Health, Witten/Herdecke University, Witten, Germany. Generative artificial intelligence tools were used solely for limited language polishing and grammar correction. The authors take full responsibility for the content, accuracy, interpretation, and conclusions of the manuscript.</p><p/></ack><notes><sec><title>Funding</title><p>This study is supported by internal departmental resources only. No external funding was received for the design of the study, data collection, data analysis, interpretation of the data, or writing of the manuscript.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: L&#x00D6; CB DG FB JE TS; Data curation: L&#x00D6; CB TS; Formal analysis: L&#x00D6; CB TS; Funding acquisition: TS; Investigation L&#x00D6;; Methodology: L&#x00D6; CB; Project administration: L&#x00D6; CB DG FB JE TS; Resources: L&#x00D6; CB TS; Software: L&#x00D6; CB TS; Supervision: L&#x00D6; CB TS; Validation: L&#x00D6; CB TS; Visualization: L&#x00D6; CB; Writing &#x2013; original draft L&#x00D6; CB TS; Writing &#x2013; review &#x0026; editing L&#x00D6; CB DG FB JE TS.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AI</term><def><p>artificial intelligence</p></def></def-item><def-item><term id="abb2">CME</term><def><p>continuing medical education</p></def></def-item><def-item><term id="abb3">CONSORT</term><def><p>Consolidated Standards of Reporting Trials</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">OCR</term><def><p>optical character recognition</p></def></def-item><def-item><term id="abb6">OSF</term><def><p>Open Science Framework</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gilson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Safranek</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>How does ChatGPT perform on the United States Medical Licensing Examination (USMLE)? The implications of large language models for medical education and knowledge assessment</article-title><source>JMIR Med Educ</source><year>2023</year><month>02</month><day>8</day><volume>9</volume><fpage>e45312</fpage><pub-id pub-id-type="doi">10.2196/45312</pub-id><pub-id pub-id-type="medline">36753318</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Watari</surname><given-names>T</given-names> </name><name name-style="western"><surname>Takagi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sakaguchi</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Performance comparison of ChatGPT-4 and Japanese medical residents in the general medicine in-training examination: comparison study</article-title><source>JMIR Med Educ</source><year>2023</year><month>12</month><day>6</day><volume>9</volume><fpage>e52202</fpage><pub-id pub-id-type="doi">10.2196/52202</pub-id><pub-id pub-id-type="medline">38055323</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ali</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>OY</given-names> </name><name name-style="western"><surname>Connolly</surname><given-names>ID</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT and GPT-4 on neurosurgery written board examinations</article-title><source>Neurosurgery</source><year>2023</year><month>12</month><day>1</day><volume>93</volume><issue>6</issue><fpage>1353</fpage><lpage>1365</lpage><pub-id pub-id-type="doi">10.1227/neu.0000000000002632</pub-id><pub-id pub-id-type="medline">37581444</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Riedel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kaefinger</surname><given-names>K</given-names> </name><name name-style="western"><surname>Stuehrenberg</surname><given-names>A</given-names> </name><etal/></person-group><article-title>ChatGPT&#x2019;s performance in German OB/GYN exams - paving the way for AI-enhanced medical education and clinical practice</article-title><source>Front Med (Lausanne)</source><year>2023</year><volume>10</volume><fpage>1296615</fpage><pub-id pub-id-type="doi">10.3389/fmed.2023.1296615</pub-id><pub-id pub-id-type="medline">38155661</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burisch</surname><given-names>C</given-names> </name><name name-style="western"><surname>Bellary</surname><given-names>A</given-names> </name><name name-style="western"><surname>Breuckmann</surname><given-names>F</given-names> </name><etal/></person-group><article-title>ChatGPT-4 performance on German continuing medical education-friend or foe (trick or treat)? Protocol for a randomized controlled trial</article-title><source>JMIR Res Protoc</source><year>2025</year><month>02</month><day>6</day><volume>14</volume><fpage>e63887</fpage><pub-id pub-id-type="doi">10.2196/63887</pub-id><pub-id pub-id-type="medline">39913914</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meo</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Alotaibi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Meo</surname><given-names>MZ</given-names> </name><name name-style="western"><surname>Meo</surname><given-names>MO</given-names> </name><name name-style="western"><surname>Hamid</surname><given-names>M</given-names> </name></person-group><article-title>Medical knowledge of ChatGPT in public health, infectious diseases, COVID-19 pandemic, and vaccines: multiple choice questions examination based performance</article-title><source>Front Public Health</source><year>2024</year><volume>12</volume><fpage>1360597</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2024.1360597</pub-id><pub-id pub-id-type="medline">38711764</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhui</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yhap</surname><given-names>N</given-names> </name><name name-style="western"><surname>Liping</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Impact of large language models on medical education and teaching adaptations</article-title><source>JMIR Med Inform</source><year>2024</year><month>07</month><day>25</day><volume>12</volume><fpage>e55933</fpage><pub-id pub-id-type="doi">10.2196/55933</pub-id><pub-id pub-id-type="medline">39087590</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zong</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Cha</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Large language models in worldwide medical exams: platform development and comprehensive analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>12</month><day>27</day><volume>26</volume><fpage>e66114</fpage><pub-id pub-id-type="doi">10.2196/66114</pub-id><pub-id pub-id-type="medline">39729356</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aster</surname><given-names>A</given-names> </name><name name-style="western"><surname>Laupichler</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Rockwell-Kollmann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Masala</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bala</surname><given-names>E</given-names> </name><name name-style="western"><surname>Raupach</surname><given-names>T</given-names> </name></person-group><article-title>ChatGPT and other large language models in medical education - scoping literature review</article-title><source>Med Sci Educ</source><year>2024</year><volume>35</volume><issue>1</issue><fpage>555</fpage><lpage>567</lpage><pub-id pub-id-type="doi">10.1007/s40670-024-02206-6</pub-id><pub-id pub-id-type="medline">40144083</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mavrych</surname><given-names>V</given-names> </name><name name-style="western"><surname>Yousef</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Yaqinuddin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bolgova</surname><given-names>O</given-names> </name></person-group><article-title>Large language models in medical education: a comparative cross-platform evaluation in answering histological questions</article-title><source>Med Educ Online</source><year>2025</year><month>12</month><volume>30</volume><issue>1</issue><fpage>2534065</fpage><pub-id pub-id-type="doi">10.1080/10872981.2025.2534065</pub-id><pub-id pub-id-type="medline">40651009</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kasagga</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sapkota</surname><given-names>A</given-names> </name><name name-style="western"><surname>Changaramkumarath</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT and large language models on medical licensing exams worldwide: a systematic review and network meta-analysis with meta-regression</article-title><source>Cureus</source><year>2025</year><month>10</month><volume>17</volume><issue>10</issue><fpage>e94300</fpage><pub-id pub-id-type="doi">10.7759/cureus.94300</pub-id><pub-id pub-id-type="medline">41230320</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>W</given-names> </name></person-group><article-title>The performance of ChatGPT on medical image-based assessments and implications for medical education</article-title><source>BMC Med Educ</source><year>2025</year><month>08</month><day>23</day><volume>25</volume><issue>1</issue><fpage>1192</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-07752-0</pub-id><pub-id pub-id-type="medline">40849473</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Noy Achiron</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kagasov</surname><given-names>S</given-names> </name><name name-style="western"><surname>Neeman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Peri</surname><given-names>T</given-names> </name><name name-style="western"><surname>Fenton</surname><given-names>C</given-names> </name></person-group><article-title>Limited performance of ChatGPT-4v and ChatGPT-4o in image-based core radiology cases</article-title><source>Clin Imaging</source><year>2026</year><month>01</month><volume>129</volume><fpage>110663</fpage><pub-id pub-id-type="doi">10.1016/j.clinimag.2025.110663</pub-id><pub-id pub-id-type="medline">41265113</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="web"><article-title>(Muster-)Berufsordnung f&#x00FC;r die in Deutschland t&#x00E4;tigen &#x00C4;rztinnen und &#x00C4;rzte</article-title><source>Bundes&#x00E4;rztekammer</source><year>2024</year><access-date>2026-07-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.bundesaerztekammer.de/fileadmin/user_upload/BAEK/Themen/Recht/_Bek_BAEK_Musterberufsordnung-AE.pdf">https://www.bundesaerztekammer.de/fileadmin/user_upload/BAEK/Themen/Recht/_Bek_BAEK_Musterberufsordnung-AE.pdf</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>Fortbildung: CME Punkte sammeln &#x2013; die 10 besten Anbieter</article-title><source>praktischArzt</source><year>2025</year><access-date>2025-08-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.praktischarzt.de/arzt/weiterbildung-fortbildung/cme-punkte-sammeln/">https://www.praktischarzt.de/arzt/weiterbildung-fortbildung/cme-punkte-sammeln/</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bazzo</surname><given-names>GT</given-names> </name><name name-style="western"><surname>Lorentz</surname><given-names>GA</given-names> </name><name name-style="western"><surname>Suarez Vargas</surname><given-names>D</given-names> </name><name name-style="western"><surname>Moreira</surname><given-names>VP</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jose</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Yilmaz</surname><given-names>E</given-names> </name><name name-style="western"><surname>Magalh&#x00E3;es</surname><given-names>J</given-names> </name><name name-style="western"><surname>Castells</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ferro</surname><given-names>N</given-names> </name><name name-style="western"><surname>Silva</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Martins</surname><given-names>F</given-names> </name></person-group><article-title>Assessing the impact of OCR errors in information retrieval</article-title><source>Advances in Information Retrieval</source><year>2020</year><publisher-name>Springer</publisher-name><fpage>102</fpage><lpage>109</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-45442-5_13</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><source>OpenAI</source><access-date>2026-6-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/">https://openai.com/</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="web"><article-title>Gemini: a family of highly capable multimodal models</article-title><source>Google DeepMind</source><access-date>2025-12-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://storage.googleapis.com/deepmind-media/gemini/gemini_1_report.pdf">https://storage.googleapis.com/deepmind-media/gemini/gemini_1_report.pdf</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Fu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Kuang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><etal/></person-group><article-title>OCRBench v2: an improved benchmark for evaluating large multimodal models on visual text localization and reasoning</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 31, 2024</comment><pub-id pub-id-type="doi">10.48550/ARXIV.2501.00321</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="web"><article-title>Verordnung zur Schaffung barrierefreier Informationstechnik nach dem Behindertengleichstellungsgesetz (Barrierefreie-Informationstechnik-Verordnung &#x2013; BITV 2.0)</article-title><source>Bundesministerium der Justiz und f&#x00FC;r Verbraucherschutz</source><access-date>2025-12-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.gesetze-im-internet.de/bitv_2_0/">https://www.gesetze-im-internet.de/bitv_2_0/</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="web"><article-title>PDF techniques for WCAG 2.0</article-title><source>World Wide Web Consortium</source><access-date>2026-05-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.w3.org/TR/WCAG20-TECHS/pdf">https://www.w3.org/TR/WCAG20-TECHS/pdf</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="web"><article-title>Web Content Accessibility Guidelines (WCAG) 2.2</article-title><source>World Wide Web Consortium</source><year>2024</year><access-date>2026-05-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.w3.org/TR/WCAG22/">https://www.w3.org/TR/WCAG22/</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cheng</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nadkarni</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Mancini</surname><given-names>MB</given-names> </name><etal/></person-group><article-title>Resuscitation education science: educational strategies to improve outcomes from cardiac arrest: a scientific statement from the American Heart Association</article-title><source>Circulation</source><year>2018</year><month>08</month><day>7</day><volume>138</volume><issue>6</issue><fpage>e82</fpage><lpage>e122</lpage><pub-id pub-id-type="doi">10.1161/CIR.0000000000000583</pub-id><pub-id pub-id-type="medline">29930020</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seetharaman</surname><given-names>R</given-names> </name></person-group><article-title>Revolutionizing medical education: can ChatGPT boost subjective learning and expression?</article-title><source>J Med Syst</source><year>2023</year><month>05</month><day>9</day><volume>47</volume><issue>1</issue><fpage>61</fpage><pub-id pub-id-type="doi">10.1007/s10916-023-01957-w</pub-id><pub-id pub-id-type="medline">37160568</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>G&#x00F6;dde</surname><given-names>D</given-names> </name><name name-style="western"><surname>N&#x00F6;hl</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wolf</surname><given-names>C</given-names> </name><etal/></person-group><article-title>A SWOT (strengths, weaknesses, opportunities, and threats) analysis of ChatGPT in the medical literature: concise review</article-title><source>J Med Internet Res</source><year>2023</year><month>11</month><day>16</day><volume>25</volume><fpage>e49368</fpage><pub-id pub-id-type="doi">10.2196/49368</pub-id><pub-id pub-id-type="medline">37865883</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="web"><article-title>Empfehlungen zur &#x00E4;rztlichen Fortbildung</article-title><source>Bundes&#x00E4;rztekammer, Arbeitsgemeinschaft der deutschen &#x00C4;rztekammern</source><year>2015</year><access-date>2025-12-23</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.bundesaerztekammer.de/fileadmin/user_upload/BAEK/Themen/Aus-Fort-Weiterbildung/Fortbildung/Empfehlungen_der_Bundesaerztekammer_zur_aerztlichen_Fortbildung_14102022.pdf">https://www.bundesaerztekammer.de/fileadmin/user_upload/BAEK/Themen/Aus-Fort-Weiterbildung/Fortbildung/Empfehlungen_der_Bundesaerztekammer_zur_aerztlichen_Fortbildung_14102022.pdf</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Q</given-names> </name></person-group><article-title>ChatGPT in medical education: bibliometric and visual analysis</article-title><source>JMIR Med Educ</source><year>2025</year><month>10</month><day>7</day><volume>11</volume><fpage>e72356</fpage><pub-id pub-id-type="doi">10.2196/72356</pub-id><pub-id pub-id-type="medline">41056572</pub-id></nlm-citation></ref></ref-list></back></article>