<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id>
      <journal-title>Journal of Medical Internet Research</journal-title>
      <issn pub-type="epub">1438-8871</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v28i1e95780</article-id>
      <article-id pub-id-type="pmid">42686197</article-id>
      <article-id pub-id-type="doi">10.2196/95780</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>A Secure, Scalable Large Language Model–Based System (CIDER) for High-Throughput Clinical Data Extraction From Medical Reports: Retrospective Validation Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Steenstra</surname>
            <given-names>Ivan</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Su</surname>
            <given-names>Zhaohui</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Sikha</surname>
            <given-names>Madhu Babu</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Linzmayer</surname>
            <given-names>Robin</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Posta</surname>
            <given-names>Máté</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0002-4912-3848</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Figler</surname>
            <given-names>Aida</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0004-1279-6226</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Dobolyi</surname>
            <given-names>Zsófia</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0003-1563-1550</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Győrffy</surname>
            <given-names>Balázs</given-names>
          </name>
          <degrees>MD, Prof Dr</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <address>
            <institution>Department of Bioinformatics</institution>
            <institution>Semmelweis University</institution>
            <addr-line>Tűzoltó u. 7.</addr-line>
            <addr-line>Budapest, 1094</addr-line>
            <country>Hungary</country>
            <phone>36 30 016 4509</phone>
            <email>gyorffylab@gmail.com</email>
          </address>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5772-3766</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Institute of Molecular Life Sciences, HUN-REN Research Centre for Natural Sciences</institution>
        <addr-line>Budapest</addr-line>
        <country>Hungary</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Department of Bioinformatics</institution>
        <institution>Semmelweis University</institution>
        <addr-line>Budapest</addr-line>
        <country>Hungary</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>Institute of Transdisciplinary Discoveries</institution>
        <institution>Medical School</institution>
        <institution>University of Pécs</institution>
        <addr-line>Pécs</addr-line>
        <country>Hungary</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Balázs Győrffy <email>gyorffylab@gmail.com</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>2</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>28</volume>
      <elocation-id>e95780</elocation-id>
      <history>
        <date date-type="received">
          <day>20</day>
          <month>3</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>15</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>10</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>10</day>
          <month>6</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Máté Posta, Aida Figler, Zsófia Dobolyi, Balázs Győrffy. Originally published in the Journal of Medical Internet Research (https://www.jmir.org), 02.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on https://www.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.jmir.org/2026/1/e95780" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>A substantial proportion of clinically relevant information remains locked in unstructured narrative documents, creating a bottleneck for clinical research, biobank annotation, registry development, and real-world evidence generation. While large language models (LLMs) enable advanced clinical text mining, adoption is constrained by concerns regarding data security, multilingual performance, and reproducibility. Manual data abstraction remains predominant for registry curation and retrospective research, despite being labor intensive, costly, and prone to variability.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>We developed and validated CIDER (Clinical Data Extractor), a secure, institutionally deployable, LLM-based pipeline for automated structured data extraction from routine clinical reports. We assessed the potential utility of the system for improving the completeness of clinical research datasets.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>CIDER uses an asynchronous FastAPI-based architecture with a locally deployed vLLM inference engine running Qwen3-VL-32B-Instruct-FP8 model in an institution-controlled environment. The system was validated on 2073 real-world Hungarian-language histopathology reports (a challenging non-English setting), using a manually curated structured database as the reference standard. Seven variables were evaluated (sex, surgery year, T stage, N stage, organ, histology, and size). Extraction performance was assessed using exact-match accuracy, weighted <italic>F</italic><sub>1</sub>-scores, Cohen κ statistics, and tolerance-based agreement thresholds for tumor size. Robustness was evaluated across temperatures from 0 to 2.0, and technical reproducibility was assessed at a temperature of 0.1 across 3 independent runs.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>The validation dataset comprised stand-alone native-text PDF pathology reports originating from multiple Hungarian oncology centers. Input document length showed a median of 3926 (mean 4258, SD 1057, IQR 3490-4688) tokens, while generated outputs contained a median of 63 (mean 62.5, SD 8.2, IQR 56-66) tokens. At a temperature of 0.1, CIDER achieved near-human agreement with expert-curated reference database, with exact-match accuracies of 99.5% for sex, 98.1% for surgery year, 95.8% for organ, 95.6% for T stage, 92.4% for N stage, 87.5% for histology, and 78.1% for tumor size. Weighted <italic>F</italic><sub>1</sub>-scores ranged from 0.87 for histology to 0.995 for sex, while Cohen κ values ranged from 0.85 for N stage to 0.99 for sex. For tumor size extraction, 83.5% to 85.3% and 87.3% to 88.7% of predictions were within 5 mm and 10 mm of the manually curated values, respectively. CIDER additionally generated candidate extractions for variables omitted during manual curation, including 62.8% (713/1136) of missing T stages and 91.5% (289/316) of tumor size values. Sensitivity testing revealed high robustness, with negligible variance at temperature=0.1 and stable performance at high temperatures (temperature=2.0).</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>CIDER demonstrates that locally deployed open-weight LLMs can reliably extract structured clinical data from complex pathology reports while preserving institutional control over sensitive data. These findings support the feasibility of secure, institutionally deployable, LLM-based extraction systems for generating research-ready datasets, facilitating clinical registry development, improving dataset completeness, and enabling scalable reuse of unstructured clinical documentation.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>large language model</kwd>
        <kwd>LLM</kwd>
        <kwd>natural language processing</kwd>
        <kwd>NLP</kwd>
        <kwd>oncology</kwd>
        <kwd>biobanking</kwd>
        <kwd>histopathology</kwd>
        <kwd>health care records</kwd>
        <kwd>patient documentation</kwd>
        <kwd>free-text documents</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>The digitalization of modern health care has resulted in an unprecedented accumulation of clinical data. It is estimated that at least 80% of this information remains trapped in unstructured, narrative documents [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Clinical reports, discharge summaries, and laboratory and histopathology reports contain detailed longitudinal information that is essential not only for medical follow-up but also for prospective and retrospective research and the development of high-quality clinical registries [<xref ref-type="bibr" rid="ref3">3</xref>]. However, the inherent variability in medical linguistics—characterized by nonstandard abbreviations, complex syntax, frequent grammatical errors, and domain-specific terminology—makes automated data extraction a painstaking challenge in medical informatics [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>].</p>
      <p>Traditionally, the conversion of narrative texts into structured variables was based on manual medical record review [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. While the labor-intensive medical record review is often considered the gold standard for data accuracy, it is profoundly limited by high costs [<xref ref-type="bibr" rid="ref7">7</xref>], low throughput, and susceptibility to human fatigue, which can lead to errors in large-scale cohorts [<xref ref-type="bibr" rid="ref8">8</xref>]. Previous attempts to automate this process using rule-based natural language processing (NLP) or early machine learning models often required extensive manual feature engineering and lacked the flexibility to adapt to different medical fields or languages [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>].</p>
      <p>Early automated extraction of clinical information relied on symbolic, rule-based systems, such as MetaMap [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>] and the clinical Text Analysis and Knowledge Extraction System (cTAKES) [<xref ref-type="bibr" rid="ref13">13</xref>], which used predefined medical ontologies and regular expressions to identify clinical entities [<xref ref-type="bibr" rid="ref14">14</xref>]. While these systems achieved high precision in specific, well-defined domains, they still required labor-intensive customization and struggled with the linguistic variability and context dependence inherent in narrative medical texts [<xref ref-type="bibr" rid="ref10">10</xref>].</p>
      <p>With the advent of the transformer architecture [<xref ref-type="bibr" rid="ref15">15</xref>], there has been a significant shift toward deep learning approaches [<xref ref-type="bibr" rid="ref16">16</xref>]. Literature has increasingly demonstrated that models pretrained on large-scale medical cohorts (such as Bidirectional Encoder Representations from Transformers [BERT] for biomedical text mining [BioBERT] [<xref ref-type="bibr" rid="ref17">17</xref>] or clinical BERT [<xref ref-type="bibr" rid="ref18">18</xref>]) outperform traditional methods by capturing complex semantic relationships within clinical documentation [<xref ref-type="bibr" rid="ref19">19</xref>]. Recent studies have successfully applied these models to extract tumor node metastasis stages, histology, and biomarker status from pathology reports in English-centric datasets [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. However, despite the fact that only a minority of cases are from English-speaking countries, research focusing on non-English clinical records remains relatively sparse [<xref ref-type="bibr" rid="ref23">23</xref>].</p>
      <p>The emergence of large language models (LLMs) represents a paradigm shift in clinical NLP. These models, trained on massive corpora, exhibit advanced contextual inference capabilities and a high degree of “zero-shot” or “few-shot” adaptability to specialized domains [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>]. Despite their potential, the adoption of LLMs in clinical environments remains hindered by 3 critical barriers: data security, multilingual performance, and reproducibility [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. The use of commercial, cloud-based LLM services would necessitate the transfer of sensitive health information to external companies and servers, raising significant ethical and legal concerns regarding patient privacy and General Data Protection Regulation (GDPR) compliance [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. Additionally, while frontier models show high proficiency in English, their performance often drops when processing other languages [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], particularly within specialized clinical domains such as oncology.</p>
      <p>To bridge this gap, we present CIDER (Clinical Data Extractor), an end-to-end system designed for the secure, high-throughput extraction of structured clinical data. CIDER is designed to transform unstructured clinical reports in real time into standardized datasets ready for statistical analysis. The schema-guided framework was designed to support flexible extraction of heterogeneous variables without task-specific retraining. The aim of the present study was to validate the performance of CIDER for extracting clinically relevant information from Hungarian-language oncological histopathology reports and to assess the feasibility of integrating institutionally deployable open-weight LLMs into clinical data curation workflows for research and registry development.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>CIDER System Architecture</title>
        <p>The CIDER framework is established as a high-throughput, secure clinical data extraction pipeline, developed primarily in Python using the FastAPI framework. To ensure strict compliance with medical data privacy, ethical standards, and full data sovereignty, the system is designed to operate entirely within an on-premises, air-gapped institutional environment. The backend uses asynchronous request handling to maintain high availability and rapid response times, which is essential for processing comprehensive document batches.</p>
        <p>The system is deployed on local institutional servers equipped with NVIDIA A40 graphics processing units (GPUs). The core extraction engine uses the open-weight Qwen3-VL-32B-Instruct-FP8 model [<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref35">35</xref>], a 32 billion parameter vision-language model selected because it combines multilingual support, strong instruction-following capabilities, and compatibility with institutionally deployable inference workflows. Model selection was based primarily on deployment feasibility and multilingual applicability rather than formal benchmarking against alternative architectures within this study. Although the A40 architecture does not provide native FP8 acceleration, the reduced-precision representation decreased memory requirements and enabled practical deployment of the 32B-parameter model within the available institutional infrastructure. Recent large-scale studies evaluating reduced-precision LLM inference have further shown that FP8-based representations can preserve model quality with minimal or negligible degradation relative to higher-precision formats while improving computational efficiency [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>].</p>
        <p>Model inference is managed via the vLLM [<xref ref-type="bibr" rid="ref38">38</xref>] serving framework, which exposes an OpenAI-compatible API. This architecture enables efficient memory use while supporting high-throughput, concurrent processing of clinical documents.</p>
        <p>The file processing workflow is fully asynchronous to maximize throughput. Upon upload, PDF documents can be processed individually or grouped based on patient-specific metadata when multi-document analysis is required. Text extraction is performed using the <italic>pymupdf</italic> (fitz) library, which parses the unstructured PDF content into machine-readable text. Each file cluster is treated as an independent processing unit.</p>
        <p>The final extraction stage integrates prompt assembly with schema-driven generation. Predefined and user-defined data columns are transformed into a rigorous schema, which works as a structural constraint. Starting from a medical record as an input, the model’s output is streamed back to the user interface in a structured JSON format for immediate clinical or research application (<xref rid="figure1" ref-type="fig">Figure 1</xref>).</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>The CIDER (Clinical Data Extractor) system architecture and data extraction workflow. Unstructured medical records in PDF format are processed through a central server. Clinical variables and model parameters (eg, temperature) can be defined, and the system handles data processing, schema-guided extraction, and structured output definition. The final output is a structured, research-ready tabular Excel table.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e95780_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>For the inference layer, the <italic>vllm</italic> library is used as the high-performance serving engine. The web application and API management are handled by <italic>fastapi</italic> package. Extraction logic is implemented using the <italic>langchain</italic> framework. For document processing, the <italic>pymupdf</italic> (fitz) library is used.</p>
        <p>In the evaluated configuration, all components involved in clinical processing—including document parsing, prompt generation, model inference, and structured output generation—were executed on institution-controlled hardware without transmitting clinical documents or extracted outputs to external cloud services or third-party APIs. The LLM inference engine, FastAPI backend, and supporting processing components were hosted on local GPU servers, and the validation dataset remained within the institutional infrastructure throughout processing. The publicly accessible URL serves as a user-facing interface, while all document processing and model inference were performed on institution-controlled infrastructure. Uploaded documents were processed within temporary storage and removed after extraction. Operational logs contained metadata only (eg, time stamps, session identifiers, filenames, token counts, processing duration, and error information), while report text bodies and extracted clinical values were not stored in operational logging systems. No external cloud-based clinical processing APIs or third-party telemetry services were used during runtime inference.</p>
      </sec>
      <sec>
        <title>Schema-Guided Extraction and Adaptive Prompting</title>
        <p>The precision of the CIDER system relies on a multilayered prompt assembly strategy designed to ensure high clinical accuracy and consistency. The extraction pipeline uses a hierarchical prompt structure that combines general extraction rules with project-specific logic and dynamic user constraints.</p>
        <p>The backbone of the application is the CIDER’s system prompt (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), which serves as a robust clinical framework that enforces medical accuracy, consistency, and standard terminology across diverse document types. The system dynamically injects user-defined column descriptions (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>) into the prompt. These descriptions are treated as authoritative rules, allowing the system to adapt its extraction logic based on the specific requirements of a research project without modifying the underlying code.</p>
        <p>In this study, both the system prompt and dynamically inserted schema definitions were written in English, whereas the input pathology reports were written in Hungarian. English prompts were selected because contemporary multilingual LLMs are predominantly trained and instruction-tuned on English-centric corpora, and previous work suggests that English prompting may support robust cross-lingual reasoning [<xref ref-type="bibr" rid="ref39">39</xref>]. Recent evidence suggests that although prompts translated into the target task language may provide marginally better multilingual robustness than English prompts, English prompts remain practically attractive in multilingual environments [<xref ref-type="bibr" rid="ref40">40</xref>]. As prompt language optimization was not systematically evaluated in the present work, English prompting should be interpreted as an implementation choice rather than an optimized design decision.</p>
      </sec>
      <sec>
        <title>OnkoBank Database Setup Validation</title>
        <p>The validation of CIDER was performed using a high-quality, manually curated dataset derived from the institutional OnkoBank database. The OnkoBank cohort included pathology reports originating from several Hungarian oncology treatment centers, thereby introducing variability in reporting practices, terminology, document structure, and writing style. This cohort consists of histopathological reports from 2073 patients who underwent surgical resection for confirmed or clinically suspected malignant neoplasms. Eligible cases included consecutive pathology reports available within the OnkoBank registry between 2022 and 2025. Reports containing native text suitable for automated processing were included in the validation cohort. All medical records related to tissue specimens were collected in connection with surgical procedures performed between 2022 and 2025.</p>
        <p>The database serves as a rigorous gold standard due to its 2-stage expert validation process. Initially, the histopathological evaluation and the subsequent medical reports were generated by board-certified pathologists specializing in the respective oncological fields. Following the issuance of these formal medical records, a secondary team of clinical experts at the department performed a comprehensive manual data mining process. This involved the systematic extraction of key clinical and pathological variables into a structured database format. No formal sample size calculation was performed. Instead, all eligible pathology reports available within the OnkoBank database during the study period were included to maximize the precision of performance estimates and reflect real-world institutional practice.</p>
      </sec>
      <sec>
        <title>Comparative Evaluation Framework and Stochastic Sensitivity Testing</title>
        <p>To validate the clinical utility of CIDER, we performed a head-to-head comparison between the automated LLM extractions and the curated database. The evaluation focused on 7 key clinical parameters: sex, T stage, N stage, primary tumor organ, histology, year of surgery, and tumor size.</p>
        <p>Before statistical comparison, both datasets underwent terminology normalization using predefined mapping rules established by a clinical expert independent of the manual data extraction process. Stages (T and N), sex, histology, and tumor organs were mapped to a unified English nomenclature. Organ-level mappings were generally direct, whereas histological entities required additional harmonization because reporting terminology frequently differed in specificity and granularity. The normalization process was intentionally conservative; subtype-level diagnoses were not collapsed into broader categories during evaluation. Consequently, more specific model outputs (eg, pancreatic ductal adenocarcinoma) were considered discordant from broader manually curated labels (eg, adenocarcinoma), even when a hierarchical relationship existed between the 2 terms.</p>
        <p>The primary metric for performance was extraction accuracy, defined as the percentage of identical values between the manually established database and the CIDER output, excluding cases where manual data were unavailable. In addition to concordance estimates, weighted <italic>F</italic><sub>1</sub>-scores together with Cohen κ statistics were calculated. For tumor size, tolerance-based agreement thresholds were evaluated.</p>
        <p>To assess the impact of model stochasticity on extraction fidelity, the analysis was performed across a range of temperature settings (temperature=0, 0.1, 0.2, 0.5, 1.0, and 2.0). Furthermore, to evaluate the technical reproducibility of the system, the extraction was repeated 3 independent times at the baseline temperature of 0.1.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>Ethics approval was obtained from the Regional and Institutional Research Ethics Committee of Semmelweis University, Budapest, Hungary (88/2021 [2021] and 88-1/2021 [2023]) and the National Centre for Public Health and Pharmacy, Hungary (NNGYK/80409-2/2025).</p>
        <p>Written informed consent was obtained from all individual participants included in the study. All patient data were processed locally within an institutionally controlled, secure environment in strict compliance with GDPR guidelines to ensure privacy and confidentiality. No compensation was provided to participants.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Descriptive Characteristics of the Validation Cohort</title>
        <p>The validation cohort comprised 2073 histopathology records, with a slightly female-predominant distribution. The temporal distribution of surgical procedures primarily spanned the years 2023 to 2024, accounting for 72.2% (1498/2073) of the total dataset. The database reflects a high-complexity surgical population; among records with available staging, 42.4% (397/937) presented with locally advanced disease (T3-T4), and 32.7% (159/485) demonstrated lymph node involvement (N1-N3). Notably, the manual “gold standard” database contained significant gaps in clinical staging, with T and N stages not reported in 54.8% (1136/2073) and 76.6% (1588/2073) of records, respectively (<xref ref-type="table" rid="table1">Table 1</xref>; <xref rid="figure2" ref-type="fig">Figure 2</xref>).</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Baseline characteristics of the OnkoBank validation cohort (N=2073).</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="370"/>
            <col width="0"/>
            <col width="600"/>
            <thead>
              <tr valign="top">
                <td colspan="3">Feature and category</td>
                <td>Records, n (%)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="4">
                  <bold>Sex</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Not reported</td>
                <td colspan="2">64 (3.1)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Female</td>
                <td colspan="2">1112 (53.6)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Male</td>
                <td colspan="2">897 (43.3)</td>
              </tr>
              <tr valign="top">
                <td colspan="4">
                  <bold>Year of surgery</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Not reported</td>
                <td colspan="2">6 (0.3)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>2022</td>
                <td colspan="2">203 (9.8)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>2023</td>
                <td colspan="2">778 (37.5)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>2024</td>
                <td colspan="2">720 (34.7)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>2025</td>
                <td colspan="2">366 (17.7)</td>
              </tr>
              <tr valign="top">
                <td colspan="4">
                  <bold>T stage</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Not reported</td>
                <td colspan="2">1136 (54.8)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>T0</td>
                <td colspan="2">28 (1.4)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>T1</td>
                <td colspan="2">312 (15)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>T2</td>
                <td colspan="2">200 (9.6)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>T3</td>
                <td colspan="2">335 (16.2)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>T4</td>
                <td colspan="2">62 (3)</td>
              </tr>
              <tr valign="top">
                <td colspan="4">
                  <bold>N stage</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Not reported</td>
                <td colspan="2">1588 (76.6)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>N0</td>
                <td colspan="2">326 (15.7)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>N1</td>
                <td colspan="2">120 (5.8)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>N2</td>
                <td colspan="2">38 (1.8)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>N3</td>
                <td colspan="2">1 (&#60;0.1)</td>
              </tr>
              <tr valign="top">
                <td colspan="4">
                  <bold>Tumor size<sup>a</sup></bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Not reported</td>
                <td colspan="2">316 (15.2)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>Mean was 36.7 (SD 27.2) mm.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Distribution of tumor locations and histology in the validation cohort. CNS: central nervous system; PitNET: pituitary neuroendocrine tumor.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e95780_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>The processed pathology reports consisted of stand-alone PDF documents. Each report corresponded to a single tumor entity of clinical interest, and no multi-document concatenation or patient-level grouping was performed during validation. All evaluated PDFs contained native text, and no optical character recognition pipeline was required during processing. Document structure and formatting varied between contributing institutions and clinical departments. Input document length showed a median of 3926 (mean 4258, SD 1057) tokens, while generated outputs contained a median of 63 (mean 62.5, SD 8.2) tokens. The mean processing time was 14.0 (SD 3.2) seconds per report (median 13.8 seconds), resulting in a total runtime of 2948 seconds for the complete validation cohort. During validation, no invalid JSON outputs, formatting errors, or processing failures were observed.</p>
      </sec>
      <sec>
        <title>CIDER</title>
        <p>The CIDER platform [<xref ref-type="bibr" rid="ref41">41</xref>] provides a streamlined, web-based interface for high-throughput clinical data extraction. The system supports batch processing of up to 1000 PDF documents in a single session. The results are exported as a standardized Excel spreadsheet. Users can interact with the system through an intuitive dashboard that allows for the selection of predefined variables. Users can also define and manage custom extraction variables, for which they must specify a data type (text, numeric, or binary), and a description of the extraction rule, with optional additional parameters (eg, regex pattern constraints for text output).</p>
        <p>The validation workflow reported in this study used text-only inputs extracted from native-text PDF files, and Qwen3-VL-32B-Instruct was selected because the broader CIDER platform was designed to support both text-based and multimodal document processing workflows.</p>
        <p>The user interface enables granular control over the extraction process through adjustable model parameters, including temperature, top-k, and nucleus sampling (top-p). An additional feature of the CIDER architecture is the inclusion of an extra system prompt, which provides an additional layer of flexibility and allows researchers to provide high-priority instructions. The system also includes a document grouping feature that can automatically concatenate files based on patient-specific prefixes, thereby allowing multi-document records to be analyzed as a single cohesive unit.</p>
        <p>The framework supports concurrent processing of multiple uploaded documents and scalable batch processing of large document collections, enabling the analysis of thousands of documents within a single workflow using up to 10 parallel processing threads per session. Structured outputs generation can be constrained through grammar-constrained decoding. To increase robustness, generated JSON outputs undergo automatic parsing and validation; if invalid formatting is detected, regeneration is attempted up to 3 times before the document is skipped. The platform additionally supports multimodal document processing workflows through a vision-language pipeline for image-based or scanned clinical documents.</p>
        <p>A critical component of the CIDER is its ability to process multilingual clinical documents. Although the validation cohort consisted of Hungarian-language pathology records, the underlying LLM supports cross-lingual extraction and generation of standardized English outputs. The system also incorporates missing data management logic, providing the option to return “NA” values in the table for missing variables.</p>
      </sec>
      <sec>
        <title>Technical Reliability and Sensitivity to Model Stochasticity</title>
        <p>The stability of the CIDER extraction pipeline was evaluated across a range of model temperatures (temperature=0 to temperature=2.0) and through repeated independent runs at temperature=0.1 (<xref ref-type="table" rid="table2">Table 2</xref>). At the baseline temperature of 0.1, the system demonstrated high reproducibility; categorical variables such as sex (99.50% accuracy) and N stage (92.37% accuracy) showed zero variance across 3 independent repetitions. Minor fluctuations were observed in more linguistically complex fields, such as histology (mean 87.51%, SD 0.05%) and T stage (mean 95.55%, SD 0.06%), though the SD remained negligible, confirming the system’s suitability for high-throughput clinical data extraction workflows.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Automated extraction accuracy across model temperatures. The table displays the concordance (%) between the manual gold standard and CIDER extractions for 7 clinical variables across a gradient of temperature settings (temperature=0 to temperature=2.0).</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="150"/>
            <col width="120"/>
            <col width="120"/>
            <col width="120"/>
            <col width="120"/>
            <col width="120"/>
            <col width="120"/>
            <col width="130"/>
            <thead>
              <tr valign="top">
                <td>Temperature</td>
                <td>Organ (n=1886)</td>
                <td>Histology (n=2004)</td>
                <td>Sex (n=2009)</td>
                <td>Year (n=2067)</td>
                <td>T stage (n=937)</td>
                <td>N stage (n=485)</td>
                <td>Size (n=1757)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>0, % (n)</td>
                <td>95.8 (1806)</td>
                <td>87.6 (1753)</td>
                <td>99.5 (1999)</td>
                <td>98.1 (2027)</td>
                <td>95.5 (895)</td>
                <td>92.4 (448)</td>
                <td>78 (1370)</td>
              </tr>
              <tr valign="top">
                <td>0.1 (average of 3), mean %</td>
                <td>95.8</td>
                <td>87.5</td>
                <td>99.5</td>
                <td>98.1</td>
                <td>95.6</td>
                <td>92.4</td>
                <td>78.1</td>
              </tr>
              <tr valign="top">
                <td>0.2, % (n)</td>
                <td>96.1 (1813)</td>
                <td>88.3 (1769)</td>
                <td>99.4 (1998)</td>
                <td>98.1 (2027)</td>
                <td>95.5 (895)</td>
                <td>91.3 (443)</td>
                <td>79.5 (1397)</td>
              </tr>
              <tr valign="top">
                <td>0.5, % (n)</td>
                <td>95.7 (1804)</td>
                <td>88.1 (1765)</td>
                <td>99.5 (1998)</td>
                <td>98 (2025)</td>
                <td>95.5 (895)</td>
                <td>92.2 (447)</td>
                <td>78.3 (1375)</td>
              </tr>
              <tr valign="top">
                <td>1.0, % (n)</td>
                <td>95.7 (1804)</td>
                <td>88.3 (1769)</td>
                <td>99.4 (1998)</td>
                <td>97.9 (2024)</td>
                <td>95.6 (896)</td>
                <td>92.4 (448)</td>
                <td>78.9 (1387)</td>
              </tr>
              <tr valign="top">
                <td>2.0, % (n)</td>
                <td>95.2 (1795)</td>
                <td>87.9 (1761)</td>
                <td>99.4 (1998)</td>
                <td>97.6 (2018)</td>
                <td>95.2 (892)</td>
                <td>92.2 (447)</td>
                <td>77.4 (1359)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>The model exhibited high robustness even at extreme temperatures across the evaluated temperature range, suggesting that extraction was driven primarily by clinically grounded textual evidence and schema constraints rather than stochastic generation effects.</p>
        <p>Notably, performance for histology and size peaked at temperature=0.2 (1769/2004, 88.27% and 1397/1757, 79.51%, respectively), suggesting that a marginal increase in sampling diversity may slightly improve the parsing of highly complex, narrative morphological descriptions. Given that temperature between the temperature=0 and temperature=0.2 interval provided a consistent result, these settings may be suitable for routine high-throughput institutional deployment (<xref ref-type="table" rid="table2">Table 2</xref>; <xref rid="figure3" ref-type="fig">Figure 3</xref>).</p>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Impact of model temperature on clinical data extraction accuracy. Clustered bar chart illustrates the concordance between the manual gold standard and CIDER (Clinical Data Extractor) outputs across 7 clinical variables. Performance is evaluated across a temperature gradient from temperature=0 (deterministic) to temperature=2.0 (high stochasticity). Accuracy starts at 70% to provide a more granular view of the results.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e95780_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>The most challenging variable was the maximum tumor diameter (tumor size in mm), which achieved a similar match rate of 77.4% (1359/1757) to 79.5% (1397/1757). Discrepancies in this category were primarily attributed to the high linguistic complexity of histological descriptions, where the computational and manual processing occasionally prioritized different measurements (eg, invasive component size vs total lesion size) when multiple diameters were listed in a single report (<xref ref-type="table" rid="table2">Table 2</xref>; <xref rid="figure3" ref-type="fig">Figure 3</xref>).</p>
        <p>Additional agreement analysis demonstrated consistently high weighted <italic>F</italic><sub>1</sub>-scores and Cohen κ statistics across variables. At the default deployment setting (temperature=0.1), weighted <italic>F</italic><sub>1</sub>-scores ranged from 0.87 for histology to 0.995 for sex, while Cohen κ values ranged from 0.85 for N stage to 0.99 for sex, indicating substantial-to-near-perfect agreement with the expert-curated reference database. For tumor size extraction, 83.5% (1467/1757) to 85.3% (1498/1757) and 87.3% (1533/1757) to 88.7% (1558/1757) of predictions fell within ±5 mm and ±10 mm of the manually curated values, respectively. Detailed extended evaluation metrics are presented in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p>
      </sec>
      <sec>
        <title>Enhanced Dataset Completeness Through Automated Extraction</title>
        <p>Beyond achieving high concordance with existing records, CIDER demonstrated a substantial capacity to recover clinically relevant data points that were omitted during the manual curation process. The manual database encompassed substantial gaps, particularly in pathological staging, where T and N stages were missing in 54.8% (1136/2073) and 76.6% (1588/2073) of cases, respectively.</p>
        <p>CIDER was able to deliver valid values for 62.8% (713/1136) of missing T stages and 10.3% (164/1588) of missing N stages (<xref ref-type="table" rid="table3">Table 3</xref>; <xref rid="figure4" ref-type="fig">Figure 4</xref>). Remarkably, the system also closed data gaps for tumor size and primary organ location, retrieving 289 and 182 additional data points, respectively (<xref ref-type="table" rid="table3">Table 3</xref>). These findings suggest that automated extraction may help identify clinically relevant information that remains uncaptured during manual curation workflows, particularly in variables with high rates of missingness. However, these candidate extractions were not included in accuracy calculations and were not independently validated against an external reference standard. Therefore, they should be interpreted as candidate annotations rather than verified extraction results, and independent confirmation would be required to before their use in clinical or research applications. Additional confusion matrices for organ classification, histology, year of surgery, and tumor size are presented in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Gap analysis and data recovery metrics: comparison of data completeness between manual curation and automated extraction using Clinical Data Extractor.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="200"/>
            <col width="300"/>
            <col width="300"/>
            <col width="200"/>
            <thead>
              <tr valign="top">
                <td>Variable</td>
                <td>Manual missing, n</td>
                <td>AI recovered, n</td>
                <td>Recovery (%)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Organ</td>
                <td>187</td>
                <td>182</td>
                <td>97.3</td>
              </tr>
              <tr valign="top">
                <td>Histology</td>
                <td>69</td>
                <td>42</td>
                <td>60.9</td>
              </tr>
              <tr valign="top">
                <td>Sex</td>
                <td>64</td>
                <td>64</td>
                <td>100</td>
              </tr>
              <tr valign="top">
                <td>Year</td>
                <td>6</td>
                <td>6</td>
                <td>100</td>
              </tr>
              <tr valign="top">
                <td>T stage</td>
                <td>1136</td>
                <td>713</td>
                <td>62.8</td>
              </tr>
              <tr valign="top">
                <td>N stage</td>
                <td>1588</td>
                <td>164</td>
                <td>10.3</td>
              </tr>
              <tr valign="top">
                <td>Size</td>
                <td>316</td>
                <td>289</td>
                <td>91.5</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>Concordance matrices highlighting the recovered data. Heat maps illustrate the agreement between manual gold standard curation and automated extraction by CIDER (Clinical Data Extractor) for (A) sex, (B) T stage, and (C) N stage. “N/A” values represent missing or nonreported data. The high density along the diagonal confirms exact-match reliability, while the “N/A” row or column intersection reveals the system’s “recovery” capacity and its “refusal” accuracy, where both human and AI correctly identified the absence of reportable information.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e95780_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Results</title>
        <p>In this retrospective validation study, we developed and evaluated CIDER as a comprehensive data extraction pipeline that can be integrated into institutional workflows. The system bridges the gap between unstructured clinical documentation and research-ready datasets by allowing investigators to define extraction targets dynamically without requiring task-specific model retraining or manual feature engineering. This design substantially lowers the technical barrier for large-scale clinical data reuse and supports rapid hypothesis generation across diverse oncological research questions [<xref ref-type="bibr" rid="ref42">42</xref>]. CIDER achieved high agreement with an expert-curated reference database across clinically relevant variables, particularly for sex, surgery year, primary tumor organ, and pathological T and N stages. The system also demonstrated high technical reproducibility across repeated runs and robustness across a broad range of temperature settings.</p>
      </sec>
      <sec>
        <title>Interpretation and Comparison With Literature</title>
        <p>Ethical issues related to the deployment of LLMs in health care remain a central concern, particularly with respect to data privacy, governance, and regulatory compliance [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. Although contemporary “frontier” models such as GPT-5 (OpenAI), Claude (Anthropic), Gemini (Google DeepMind), and Grok (xAI) define the upper bound of performance across general reasoning and instruction-following tasks, their closed-source nature and reliance on third-party, cloud-based deployment necessitate the transfer of sensitive protected health information to external infrastructure [<xref ref-type="bibr" rid="ref45">45</xref>]. Such practices raise substantial ethical and legal risks and challenges under strict regulatory frameworks, such as the HIPAA (Health Insurance Portability and Accountability Act) and GDPR, effectively prohibiting their use in privacy-sensitive clinical environments [<xref ref-type="bibr" rid="ref46">46</xref>]. In parallel, a rapidly expanding ecosystem of open-weight LLMs—including DeepSeek, Gemma, GPT-OSS, Mistral, and the Qwen family—has enabled transparent, institutionally deployable alternatives that preserve full data sovereignty [<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>-<xref ref-type="bibr" rid="ref49">49</xref>]. Within this landscape, and particularly in the 20 to 40 billion parameter class, Qwen3-VL-32B established a strong performance benchmark [<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref35">35</xref>], achieving state-of-the-art or near–state-of-the-art results across multimodal reasoning, instruction following, and multilingual understanding. The objective of this study was to validate an institutionally deployable LLM-based extraction pipeline rather than to perform comparative benchmarking across language models. The model’s strong multilingual performance on benchmarks such as Massive Multitask Language Understanding [<xref ref-type="bibr" rid="ref50">50</xref>], XStoryCloze [<xref ref-type="bibr" rid="ref51">51</xref>], and MMBench [<xref ref-type="bibr" rid="ref52">52</xref>] further supported its suitability for applications involving morphologically complex languages. Although formal comparisons against alternative architectures were beyond the scope of this work, future benchmarking studies will be important for determining how model selection influences extraction performance across languages and clinical domains. Furthermore, our pipeline was explicitly designed to operate within an on-premises, air-gapped institutional environment, ensuring regulatory compliance, the elimination of third-party data transfer, and full data security. This deployment strategy primarily affects governance and privacy considerations rather than the underlying extraction performance itself because the inference process remains identical irrespective of whether computation occurs locally or on externally hosted infrastructure.</p>
        <p>Rule-based and traditional clinical NLP systems remain important comparators because of their interpretability and established use in well-defined entity extraction tasks [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref14">14</xref>]. However, registry curation often requires more than detecting candidate entities. The same report may contain multiple dates, specimen dimensions, tumor measurements, historical diagnoses, molecular addenda, and anatomical references, making it challenging to identify which information corresponds to the registry variable of interest. In this setting, the main challenge is contextual selection and higher-order interpretation rather than simple pattern recognition [<xref ref-type="bibr" rid="ref10">10</xref>]. This represents a potential advantage of schema-guided LLM-based extraction, which can use the broader document context to identify the clinically relevant value without task-specific retraining [<xref ref-type="bibr" rid="ref53">53</xref>]. Nevertheless, standardized benchmarking datasets and direct comparisons with rule-based, traditional machine learning, and LLM-based approaches will be necessary to quantify the incremental value of each methodology [<xref ref-type="bibr" rid="ref54">54</xref>].</p>
        <p>CIDER achieved the highest accuracy when used for well-defined categorical and temporal variables, such as sex and year of surgery, exceeding 98% concordance with the manual reference data. These results are consistent with prior observations that structured or semistructured elements embedded within narrative text are particularly amenable to automated extraction [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. The high agreement observed for sex extraction may be partly explained by the fact that Hungarian given names are typically sex-specific, allowing the model to infer the database-level sex variable from patient names when sex was not explicitly stated in the report. This finding illustrates the ability of LLM-based extraction to use contextual information for variables that are not always presented in an explicitly structured format. Importantly, the system also demonstrated strong performance in more complex oncological variables, including T and N stages and primary tumor organ, which require contextual interpretation rather than simple keyword matching. Accuracy values above 90% for these parameters indicate that modern LLMs can reliably capture clinically meaningful relationships. Notably, many discrepancies in histology were not model failures, as CIDER often provided a more granular diagnosis (eg, specifying pancreatic ductal adenocarcinoma where the manual database only recorded adenocarcinoma), which was flagged as a mismatch despite the model providing a more specific diagnostic description.</p>
        <p>The extraction of tumor size proved to be the most challenging task, with an identical match rate of 78.1%. This finding aligns with known difficulties in both manual and automated abstraction of quantitative measurements from pathology reports, where multiple diameters may be reported for different components of a lesion (eg, total tumor size vs invasive component) [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]. This underscores a broader methodological challenge in clinical NLP evaluation: apparent disagreement may reflect ambiguity in the source text rather than model failure.</p>
        <p>The extraction performance observed in the present study is comparable to that reported in recent pathology and clinical information extraction studies using contemporary LLMs. Recent investigations have demonstrated high accuracy for extraction of pathological staging variables, histological diagnoses, and structured cancer registry elements from free-text pathology reports and other clinical documents [<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref58">58</xref>]. Similarly, locally deployed open-weight LLMs have been shown to achieve performance comparable to that of proprietary frontier models while allowing institutions to retain control over sensitive clinical data [<xref ref-type="bibr" rid="ref59">59</xref>-<xref ref-type="bibr" rid="ref61">61</xref>]. Importantly, unlike most prior studies that focused primarily on English-language clinical documentation, the present validation was performed using Hungarian pathology reports, demonstrating that high extraction accuracy can be achieved in a morphologically complex and comparatively lower-resource language setting.</p>
        <p>An important strength of CIDER is its ability to process non-English clinical documents, as demonstrated by its validation using Hungarian-language pathology reports. Most existing clinical NLP systems and benchmark datasets are heavily English-centric, limiting their applicability in any non–English-speaking health care system [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref62">62</xref>]. Validation in Hungarian is particularly relevant because Hungarian represents a morphologically complex and comparatively lower-resource language. Recent multilingual benchmark studies have demonstrated persistent performance differences between high-resource and lower-resource languages in contemporary LLMs [<xref ref-type="bibr" rid="ref63">63</xref>].</p>
        <p>The rapid evolution of LLMs represents both a challenge and an opportunity for clinical NLP research. Consequently, the primary transferable insight of the present study is not tied to a specific model version but rather to the broader feasibility of institutionally deployable, schema-driven extraction pipelines for transforming unstructured clinical narratives into structured research datasets. The present work therefore provides transferable methodological insights regarding (1) secure on-premises deployment of clinical LLM systems, (2) schema-guided extraction strategies, (3) validation of clinical NLP workflows in morphologically complex and lower-resource languages, and (4) reproducible integration of LLM-based extraction into clinical registry development pipelines. While future model generations will likely improve extraction quality, multilingual robustness, and computational efficiency, the core workflow and validation principles presented here are expected to remain applicable across evolving architectures and document types.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>This study has some limitations that should be considered when interpreting the results. Although the underlying Qwen3-VL-32B-Instruct model possesses multilingual capabilities, the present validation was restricted to Hungarian-language pathology reports within the OnkoBank registry framework; therefore, broader applicability across independent cohorts, additional institutions, other languages, and document types requires external validation. Second, while the manually curated OnkoBank database served as a high-quality gold standard, manual data extraction itself is not immune to error. Third, the candidate extractions were clinically plausible extractions not present in the manual database, and these could not be verified against an independent gold standard. Fourth, although the schema-guided architecture of CIDER was designed to support adaptation across extraction tasks, further validation across additional medical specialties and clinical workflows will be necessary to fully assess generalizability.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>In summary, CIDER is an end-to-end, institutionally deployable platform designed to enable secure, scalable, and reliable extraction of structured clinical data from unstructured pathology reports using an LLM. Our results demonstrate that contemporary LLM-based approaches can achieve high agreement with an expert-curated reference database across a range of clinically relevant variables, while simultaneously addressing long-standing limitations of manual data extraction. While objective benchmarks for LLM-based clinical information extraction are still evolving, our results suggest that institutionally deployed LLMs can meet the practical performance requirements for real-world clinical research applications. More broadly, schema-guided and privacy-preserving LLM-based extraction frameworks may facilitate the development of high-quality clinical registries and enable scalable secondary use of routinely collected health care data for research purposes.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>The CIDER (Clinical Data Extractor) system prompt.</p>
        <media xlink:href="jmir_v28i1e95780_app1.pdf" xlink:title="PDF File  (Adobe PDF File), 65 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Predefined schema and variable descriptions.</p>
        <media xlink:href="jmir_v28i1e95780_app2.pdf" xlink:title="PDF File  (Adobe PDF File), 39 KB"/>
      </supplementary-material>
      <supplementary-material id="app3">
        <label>Multimedia Appendix 3</label>
        <p>Extended evaluation metrics across temperature settings.</p>
        <media xlink:href="jmir_v28i1e95780_app3.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 9 KB"/>
      </supplementary-material>
      <supplementary-material id="app4">
        <label>Multimedia Appendix 4</label>
        <p>Confusion matrices for organ, histology, year of surgery, and tumor size extraction.</p>
        <media xlink:href="jmir_v28i1e95780_app4.pdf" xlink:title="PDF File  (Adobe PDF File), 1219 KB"/>
      </supplementary-material>
      <supplementary-material id="app5">
        <label>Multimedia Appendix 5</label>
        <p>STARD-AI checklist.</p>
        <media xlink:href="jmir_v28i1e95780_app5.docx" xlink:title="DOCX File , 18 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">BERT</term>
          <def>
            <p>Bidirectional Encoder Representations from Transformers</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">BioBERT</term>
          <def>
            <p>Bidirectional Encoder Representations from Transformers for biomedical text mining</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">CIDER</term>
          <def>
            <p>Clinical Data Extractor</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">cTAKES</term>
          <def>
            <p>clinical Text Analysis and Knowledge Extraction System</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">GDPR</term>
          <def>
            <p>General Data Protection Regulation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">GPU</term>
          <def>
            <p>graphics processing unit</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">HIPAA</term>
          <def>
            <p>Health Insurance Portability and Accountability Act</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">NLP</term>
          <def>
            <p>natural language processing</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>A GPT-based AI grammar check was used to improve the English of the manuscript.</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>This project was supported by the National Research, Development, and Innovation Office (2025-1.2.1-HU-RIZONT-2025-00011 and 2024-1.2.2-ERA_NET-2024-00015) and the Semmelweis Lendület Programme.</p>
      </sec>
    </notes>
    <notes>
      <sec>
        <title>Data Availability</title>
        <p>The clinical data analyzed in this study contain sensitive patient information and cannot be made publicly available because of institutional and regulatory restrictions. Summary data supporting the findings of this study are included within the manuscript and its supplementary materials. Requests for additional information may be directed to the corresponding author and will be considered on a case-by-case basis, subject to institutional approval and applicable data protection regulations.</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>MP: conceptualization, methodology, software, validation, formal analysis, data curation, writing—original draft, writing—review and editing, and visualization. AF: methodology, formal analysis, data curation, and writing—review and editing. ZD: methodology, software, data curation, and writing—review and editing. BG: conceptualization, validation, writing—original draft, writing—review and editing, supervision, project administration, and funding acquisition.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Murdoch</surname>
              <given-names>TB</given-names>
            </name>
            <name name-style="western">
              <surname>Detsky</surname>
              <given-names>AS</given-names>
            </name>
          </person-group>
          <article-title>The inevitable application of big data to health care</article-title>
          <source>JAMA</source>
          <year>2013</year>
          <month>04</month>
          <day>03</day>
          <volume>309</volume>
          <issue>13</issue>
          <fpage>1351</fpage>
          <lpage>2</lpage>
          <pub-id pub-id-type="doi">10.1001/jama.2013.393</pub-id>
          <pub-id pub-id-type="medline">23549579</pub-id>
          <pub-id pub-id-type="pii">1674245</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kong</surname>
              <given-names>HJ</given-names>
            </name>
          </person-group>
          <article-title>Managing unstructured big data in healthcare system</article-title>
          <source>Healthc Inform Res</source>
          <year>2019</year>
          <month>01</month>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <lpage>2</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/30788175"/>
          </comment>
          <pub-id pub-id-type="doi">10.4258/hir.2019.25.1.1</pub-id>
          <pub-id pub-id-type="medline">30788175</pub-id>
          <pub-id pub-id-type="pmcid">PMC6372467</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jensen</surname>
              <given-names>PB</given-names>
            </name>
            <name name-style="western">
              <surname>Jensen</surname>
              <given-names>LJ</given-names>
            </name>
            <name name-style="western">
              <surname>Brunak</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Mining electronic health records: towards better research applications and clinical care</article-title>
          <source>Nat Rev Genet</source>
          <year>2012</year>
          <month>05</month>
          <day>02</day>
          <volume>13</volume>
          <issue>6</issue>
          <fpage>395</fpage>
          <lpage>405</lpage>
          <pub-id pub-id-type="doi">10.1038/nrg3208</pub-id>
          <pub-id pub-id-type="medline">22549152</pub-id>
          <pub-id pub-id-type="pii">nrg3208</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cardamone</surname>
              <given-names>NC</given-names>
            </name>
            <name name-style="western">
              <surname>Olfson</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Schmutte</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Ungar</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Cullen</surname>
              <given-names>SW</given-names>
            </name>
            <name name-style="western">
              <surname>Williams</surname>
              <given-names>NJ</given-names>
            </name>
            <name name-style="western">
              <surname>Marcus</surname>
              <given-names>SC</given-names>
            </name>
          </person-group>
          <article-title>Classifying unstructured text in electronic health records for mental health prediction models: large language model evaluation study</article-title>
          <source>JMIR Med Inform</source>
          <year>2025</year>
          <month>01</month>
          <day>21</day>
          <volume>13</volume>
          <fpage>e65454</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://medinform.jmir.org/2025//e65454/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/65454</pub-id>
          <pub-id pub-id-type="medline">39864953</pub-id>
          <pub-id pub-id-type="pii">v13i1e65454</pub-id>
          <pub-id pub-id-type="pmcid">PMC11884378</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sedlakova</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Daniore</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Horn Wintsch</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wolf</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Stanikic</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Haag</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Sieber</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Schneider</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Staub</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Alois Ettlin</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Grübner</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Rinaldi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>von Wyl</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Challenges and best practices for digital unstructured data enrichment in health research: a systematic narrative review</article-title>
          <source>PLOS Digit Health</source>
          <year>2023</year>
          <month>10</month>
          <day>11</day>
          <volume>2</volume>
          <issue>10</issue>
          <fpage>e0000347</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pdig.0000347"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pdig.0000347</pub-id>
          <pub-id pub-id-type="medline">37819910</pub-id>
          <pub-id pub-id-type="pii">PDIG-D-22-00215</pub-id>
          <pub-id pub-id-type="pmcid">PMC10566734</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Hao</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>VZ</given-names>
            </name>
            <name name-style="western">
              <surname>Hollander</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Ng</surname>
              <given-names>RT</given-names>
            </name>
            <name name-style="western">
              <surname>Isaac</surname>
              <given-names>KV</given-names>
            </name>
          </person-group>
          <article-title>Automated medical chart review for breast cancer outcomes research: a novel natural language processing extraction system</article-title>
          <source>BMC Med Res Methodol</source>
          <year>2022</year>
          <month>05</month>
          <day>12</day>
          <volume>22</volume>
          <issue>1</issue>
          <fpage>136</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedresmethodol.biomedcentral.com/articles/10.1186/s12874-022-01583-z"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12874-022-01583-z</pub-id>
          <pub-id pub-id-type="medline">35549854</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12874-022-01583-z</pub-id>
          <pub-id pub-id-type="pmcid">PMC9101856</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gauthier</surname>
              <given-names>MP</given-names>
            </name>
            <name name-style="western">
              <surname>Law</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Le</surname>
              <given-names>LW</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>JJN</given-names>
            </name>
            <name name-style="western">
              <surname>Zahir</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Nirmalakumar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sung</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pettengell</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Aviv</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chu</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Sacher</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Bradbury</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Shepherd</surname>
              <given-names>FA</given-names>
            </name>
            <name name-style="western">
              <surname>Leighl</surname>
              <given-names>NB</given-names>
            </name>
          </person-group>
          <article-title>Automating access to real-world evidence</article-title>
          <source>JTO Clin Res Rep</source>
          <year>2022</year>
          <month>05</month>
          <day>17</day>
          <volume>3</volume>
          <issue>6</issue>
          <fpage>100340</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2666-3643(22)00064-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jtocrr.2022.100340</pub-id>
          <pub-id pub-id-type="medline">35719866</pub-id>
          <pub-id pub-id-type="pii">S2666-3643(22)00064-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC9201015</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zozus</surname>
              <given-names>MN</given-names>
            </name>
            <name name-style="western">
              <surname>Pieper</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>TR</given-names>
            </name>
            <name name-style="western">
              <surname>Franklin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Smith</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Factors affecting accuracy of data abstracted from medical records</article-title>
          <source>PLoS One</source>
          <year>2015</year>
          <month>10</month>
          <day>20</day>
          <volume>10</volume>
          <issue>10</issue>
          <fpage>e0138649</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pone.0138649"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0138649</pub-id>
          <pub-id pub-id-type="medline">26484762</pub-id>
          <pub-id pub-id-type="pii">PONE-D-14-29205</pub-id>
          <pub-id pub-id-type="pmcid">PMC4615628</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sheikhalishahi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Miotto</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Dudley</surname>
              <given-names>JT</given-names>
            </name>
            <name name-style="western">
              <surname>Lavelli</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Rinaldi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Osmani</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Natural language processing of clinical notes on chronic diseases: systematic review</article-title>
          <source>JMIR Med Inform</source>
          <year>2019</year>
          <month>04</month>
          <day>27</day>
          <volume>7</volume>
          <issue>2</issue>
          <fpage>e12239</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://medinform.jmir.org/2019/2/e12239/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/12239</pub-id>
          <pub-id pub-id-type="medline">31066697</pub-id>
          <pub-id pub-id-type="pii">v7i2e12239</pub-id>
          <pub-id pub-id-type="pmcid">PMC6528438</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Rastegar-Mojarad</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Moon</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Afzal</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zeng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Mehrabi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sohn</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Clinical information extraction applications: a literature review</article-title>
          <source>J Biomed Inform</source>
          <year>2018</year>
          <month>01</month>
          <volume>77</volume>
          <fpage>34</fpage>
          <lpage>49</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(17)30256-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2017.11.011</pub-id>
          <pub-id pub-id-type="medline">29162496</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(17)30256-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC5771858</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Aronson</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Lang</surname>
              <given-names>FM</given-names>
            </name>
          </person-group>
          <article-title>An overview of MetaMap: historical perspective and recent advances</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2010</year>
          <volume>17</volume>
          <issue>3</issue>
          <fpage>229</fpage>
          <lpage>36</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/20442139"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/jamia.2009.002733</pub-id>
          <pub-id pub-id-type="medline">20442139</pub-id>
          <pub-id pub-id-type="pii">17/3/229</pub-id>
          <pub-id pub-id-type="pmcid">PMC2995713</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Aronson</surname>
              <given-names>AR</given-names>
            </name>
          </person-group>
          <article-title>Effective mapping of biomedical text to the UMLS Metathesaurus: the MetaMap program</article-title>
          <source>Proc AMIA Symp</source>
          <year>2001</year>
          <fpage>17</fpage>
          <lpage>21</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/11825149"/>
          </comment>
          <pub-id pub-id-type="medline">11825149</pub-id>
          <pub-id pub-id-type="pii">D010001275</pub-id>
          <pub-id pub-id-type="pmcid">PMC2243666</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Savova</surname>
              <given-names>GK</given-names>
            </name>
            <name name-style="western">
              <surname>Masanz</surname>
              <given-names>JJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ogren</surname>
              <given-names>PV</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sohn</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kipper-Schuler</surname>
              <given-names>KC</given-names>
            </name>
            <name name-style="western">
              <surname>Chute</surname>
              <given-names>CG</given-names>
            </name>
          </person-group>
          <article-title>Mayo clinical Text Analysis and Knowledge Extraction System (cTAKES): architecture, component evaluation and applications</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2010</year>
          <volume>17</volume>
          <issue>5</issue>
          <fpage>507</fpage>
          <lpage>13</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/20819853"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/jamia.2009.001560</pub-id>
          <pub-id pub-id-type="medline">20819853</pub-id>
          <pub-id pub-id-type="pii">17/5/507</pub-id>
          <pub-id pub-id-type="pmcid">PMC2995668</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Meystre</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Savova</surname>
              <given-names>GK</given-names>
            </name>
            <name name-style="western">
              <surname>Kipper-Schuler</surname>
              <given-names>KC</given-names>
            </name>
            <name name-style="western">
              <surname>Hurdle</surname>
              <given-names>JF</given-names>
            </name>
          </person-group>
          <article-title>Extracting information from textual documents in the electronic health record: a review of recent research</article-title>
          <source>Yearb Med Inform</source>
          <year>2008</year>
          <fpage>128</fpage>
          <lpage>44</lpage>
          <pub-id pub-id-type="medline">18660887</pub-id>
          <pub-id pub-id-type="pii">me08010128</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vaswani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shazeer</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Parmar</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Uszkoreit</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jones</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Gomez</surname>
              <given-names>AN</given-names>
            </name>
            <name name-style="western">
              <surname>Kaiser</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Polosukhin</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Attention is all you need</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on June 12, 2017</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Roberts</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Datta</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Du</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ji</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Si</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Soni</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Xiang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Deep learning in clinical natural language processing: a methodical review</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2020</year>
          <month>03</month>
          <day>01</day>
          <volume>27</volume>
          <issue>3</issue>
          <fpage>457</fpage>
          <lpage>70</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31794016"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocz200</pub-id>
          <pub-id pub-id-type="medline">31794016</pub-id>
          <pub-id pub-id-type="pii">5651084</pub-id>
          <pub-id pub-id-type="pmcid">PMC7025365</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yoon</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>So</surname>
              <given-names>CH</given-names>
            </name>
            <name name-style="western">
              <surname>Kang</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title>
          <source>Bioinformatics</source>
          <year>2020</year>
          <month>02</month>
          <day>15</day>
          <volume>36</volume>
          <issue>4</issue>
          <fpage>1234</fpage>
          <lpage>40</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31501885"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id>
          <pub-id pub-id-type="medline">31501885</pub-id>
          <pub-id pub-id-type="pii">5566506</pub-id>
          <pub-id pub-id-type="pmcid">PMC7703786</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Alsentzer</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Murphy</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Boag</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Weng</surname>
              <given-names>WH</given-names>
            </name>
            <name name-style="western">
              <surname>Jindi</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Naumann</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>McDermott</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <person-group person-group-type="editor">
            <name name-style="western">
              <surname>Rumshisky</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Roberts</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Bethard</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Naumann</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>Publicly available clinical BERT embeddings</article-title>
          <source>Proceedings of the 2nd Clinical Natural Language Processing Workshop</source>
          <year>2019</year>
          <publisher-loc>Stroudsburg, PA</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>72</fpage>
          <lpage>8</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yan</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <person-group person-group-type="editor">
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Cohen</surname>
              <given-names>KB</given-names>
            </name>
            <name name-style="western">
              <surname>Ananiadou</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tsujii</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Transfer learning in biomedical natural language processing: an evaluation of BERT and ELMo on ten benchmarking datasets</article-title>
          <source>Proceedings of the 18th BioNLP Workshop and Shared Task</source>
          <year>2019</year>
          <publisher-loc>Stroudsburg, PA</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>58</fpage>
          <lpage>65</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Young</surname>
              <given-names>MT</given-names>
            </name>
            <name name-style="western">
              <surname>Qiu</surname>
              <given-names>JX</given-names>
            </name>
            <name name-style="western">
              <surname>Yoon</surname>
              <given-names>HJ</given-names>
            </name>
            <name name-style="western">
              <surname>Christian</surname>
              <given-names>JB</given-names>
            </name>
            <name name-style="western">
              <surname>Fearn</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Tourassi</surname>
              <given-names>GD</given-names>
            </name>
            <name name-style="western">
              <surname>Ramanthan</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Hierarchical attention networks for information extraction from cancer pathology reports</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2018</year>
          <month>03</month>
          <day>01</day>
          <volume>25</volume>
          <issue>3</issue>
          <fpage>321</fpage>
          <lpage>30</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/29155996"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocx131</pub-id>
          <pub-id pub-id-type="medline">29155996</pub-id>
          <pub-id pub-id-type="pii">4636780</pub-id>
          <pub-id pub-id-type="pmcid">PMC7282502</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Si</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Roberts</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Enhancing clinical concept extraction with contextual embeddings</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2019</year>
          <month>11</month>
          <day>01</day>
          <volume>26</volume>
          <issue>11</issue>
          <fpage>1297</fpage>
          <lpage>304</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31265066"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocz096</pub-id>
          <pub-id pub-id-type="medline">31265066</pub-id>
          <pub-id pub-id-type="pii">5527248</pub-id>
          <pub-id pub-id-type="pmcid">PMC6798561</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Abedian</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sholle</surname>
              <given-names>ET</given-names>
            </name>
            <name name-style="western">
              <surname>Adekkanattu</surname>
              <given-names>PM</given-names>
            </name>
            <name name-style="western">
              <surname>Cusick</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Weiner</surname>
              <given-names>SE</given-names>
            </name>
            <name name-style="western">
              <surname>Shoag</surname>
              <given-names>JE</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Campion</surname>
              <given-names>TR Jr</given-names>
            </name>
          </person-group>
          <article-title>Automated extraction of tumor staging and diagnosis information from surgical pathology reports</article-title>
          <source>JCO Clin Cancer Inform</source>
          <year>2021</year>
          <month>10</month>
          <volume>5</volume>
          <fpage>1054</fpage>
          <lpage>61</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/34694896"/>
          </comment>
          <pub-id pub-id-type="doi">10.1200/CCI.21.00065</pub-id>
          <pub-id pub-id-type="medline">34694896</pub-id>
          <pub-id pub-id-type="pmcid">PMC8812635</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Névéol</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dalianis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Velupillai</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Savova</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Zweigenbaum</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Clinical natural language processing in languages other than English: opportunities and challenges</article-title>
          <source>J Biomed Semantics</source>
          <year>2018</year>
          <month>03</month>
          <day>30</day>
          <volume>9</volume>
          <issue>1</issue>
          <fpage>12</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://jbiomedsem.biomedcentral.com/articles/10.1186/s13326-018-0179-8"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s13326-018-0179-8</pub-id>
          <pub-id pub-id-type="medline">29602312</pub-id>
          <pub-id pub-id-type="pii">10.1186/s13326-018-0179-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC5877394</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Brown</surname>
              <given-names>TB</given-names>
            </name>
            <name name-style="western">
              <surname>Mann</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Ryder</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Subbiah</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kaplan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Dhariwal</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Neelakantan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shyam</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Sastry</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Askell</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Agarwal</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Herbert-Voss</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Krueger</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Henighan</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Child</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ramesh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Ziegler</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Winter</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hesse</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sigler</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Litwin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gray</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chess</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Berner</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>McCandlish</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Radford</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sutskever</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Amodei</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Language models are few-shot learners</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on May 28, 2020</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bommasani</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Hudson</surname>
              <given-names>DA</given-names>
            </name>
            <name name-style="western">
              <surname>Adeli</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Altman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Arora</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>von Arx</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Bernstein</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Bohg</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bosselut</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Brunskill</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Brynjolfsson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Buch</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Card</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Castellon</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Chatterji</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Creel</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Davis</surname>
              <given-names>JQ</given-names>
            </name>
            <name name-style="western">
              <surname>Demszky</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Donahue</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Doumbouya</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Durmus</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Ermon</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Etchemendy</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ethayarajh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Fei-Fei</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Finn</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Gale</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Gillespie</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Goel</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Goodman</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Grossman</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Guha</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Hashimoto</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Henderson</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Hewitt</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ho</surname>
              <given-names>DE</given-names>
            </name>
            <name name-style="western">
              <surname>Hong</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Hsu</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Icard</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Jain</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jurafsky</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kalluri</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Karamcheti</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Keeling</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Khani</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Khattab</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Koh</surname>
              <given-names>PW</given-names>
            </name>
            <name name-style="western">
              <surname>Krass</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Krishna</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Kuditipudi</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Kumar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Ladhak</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Leskovec</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Levent</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>XL</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Malik</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Manning</surname>
              <given-names>CD</given-names>
            </name>
            <name name-style="western">
              <surname>Mirchandani</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Mitchell</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Munyikwa</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Nair</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Narayan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Narayanan</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Newman</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Nie</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Niebles</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Nilforoshan</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Nyarko</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ogut</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Orr</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Papadimitriou</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>Piech</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Portelance</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Potts</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Raghunathan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Reich</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Rong</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Roohani</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ruiz</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Ryan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ré</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Sadigh</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Sagawa</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Santhanam</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Shih</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Srinivasan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tamkin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Taori</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Thomas</surname>
              <given-names>AW</given-names>
            </name>
            <name name-style="western">
              <surname>Tramèr</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>RE</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>On the opportunities and risks of foundation models</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on August 16, 2021</comment>
          <pub-id pub-id-type="doi">10.48550/ARXIV.2108.07258</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>van der Loo</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>van der Valk</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>van den Broek</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Atsma</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Staring</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Scherptong</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Large language models for structured cardiovascular data extraction: a foundation for scalable research and clinical applications</article-title>
          <source>Eur Heart J Digit Health</source>
          <year>2025</year>
          <month>11</month>
          <day>14</day>
          <volume>7</volume>
          <issue>2</issue>
          <fpage>ztaf127</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/ehjdh/article-lookup/doi/10.1093/ehjdh/ztaf127"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/ehjdh/ztaf127</pub-id>
          <pub-id pub-id-type="medline">41684376</pub-id>
          <pub-id pub-id-type="pii">ztaf127</pub-id>
          <pub-id pub-id-type="pmcid">PMC12893214</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>CJ</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Suleyman</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>King</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Key challenges for delivering clinical impact with artificial intelligence</article-title>
          <source>BMC Med</source>
          <year>2019</year>
          <month>10</month>
          <day>29</day>
          <volume>17</volume>
          <issue>1</issue>
          <fpage>195</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedicine.biomedcentral.com/articles/10.1186/s12916-019-1426-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12916-019-1426-2</pub-id>
          <pub-id pub-id-type="medline">31665002</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12916-019-1426-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC6821018</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Miotto</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Dudley</surname>
              <given-names>JT</given-names>
            </name>
          </person-group>
          <article-title>Deep learning for healthcare: review, opportunities and challenges</article-title>
          <source>Brief Bioinform</source>
          <year>2018</year>
          <month>11</month>
          <day>27</day>
          <volume>19</volume>
          <issue>6</issue>
          <fpage>1236</fpage>
          <lpage>46</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/28481991"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bib/bbx044</pub-id>
          <pub-id pub-id-type="medline">28481991</pub-id>
          <pub-id pub-id-type="pii">3800524</pub-id>
          <pub-id pub-id-type="pmcid">PMC6455466</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Price</surname>
              <given-names>WN 2nd</given-names>
            </name>
            <name name-style="western">
              <surname>Cohen</surname>
              <given-names>IG</given-names>
            </name>
          </person-group>
          <article-title>Privacy in the age of medical big data</article-title>
          <source>Nat Med</source>
          <year>2019</year>
          <month>01</month>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>37</fpage>
          <lpage>43</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/30617331"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41591-018-0272-7</pub-id>
          <pub-id pub-id-type="medline">30617331</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-018-0272-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC6376961</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rieke</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Hancox</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Milletarì</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Roth</surname>
              <given-names>HR</given-names>
            </name>
            <name name-style="western">
              <surname>Albarqouni</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Bakas</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Galtier</surname>
              <given-names>MN</given-names>
            </name>
            <name name-style="western">
              <surname>Landman</surname>
              <given-names>BA</given-names>
            </name>
            <name name-style="western">
              <surname>Maier-Hein</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Ourselin</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sheller</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Summers</surname>
              <given-names>RM</given-names>
            </name>
            <name name-style="western">
              <surname>Trask</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Baust</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Cardoso</surname>
              <given-names>MJ</given-names>
            </name>
          </person-group>
          <article-title>The future of digital health with federated learning</article-title>
          <source>NPJ Digit Med</source>
          <year>2020</year>
          <month>09</month>
          <day>14</day>
          <volume>3</volume>
          <fpage>119</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-020-00323-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-020-00323-1</pub-id>
          <pub-id pub-id-type="medline">33015372</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-020-00323-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC7490367</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ruder</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Siddhant</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Neubig</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Firat</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <person-group person-group-type="editor">
            <name name-style="western">
              <surname>Daumé</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>XTREME: a massively multilingual multi-task benchmark for evaluating cross-lingual generalization</article-title>
          <source>ICML'20: Proceedings of the 37th International Conference on Machine Learning</source>
          <year>2020</year>
          <publisher-loc>Norfolk, MA</publisher-loc>
          <publisher-name>JMLR.org</publisher-name>
          <fpage>4411</fpage>
          <lpage>21</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Pava</surname>
              <given-names>JN</given-names>
            </name>
            <name name-style="western">
              <surname>Meinhardt</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zaman</surname>
              <given-names>HB</given-names>
            </name>
            <name name-style="western">
              <surname>Friedman</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Truong</surname>
              <given-names>ST</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Cryst</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Marivate</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Koyejo</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Mind the (language) gap: mapping the challenges of LLM development in low-resource language contexts</article-title>
          <source>HAI Stanford University</source>
          <year>2025</year>
          <access-date>2026-08-07</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://hai.stanford.edu/policy/mind-the-language-gap-mapping-the-challenges-of-llm-development-in-low-resource-language-contexts">https://hai.stanford.edu/policy/mind-the-language-gap-mapping-the-challenges-of-llm-development-in-low-resource-language-contexts</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cai</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Deng</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ding</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Ge</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Ge</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hui</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Lv</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Men</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Meng</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zhong</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Qwen3-VL technical report</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on November 26, 2025</comment>
          <pub-id pub-id-type="doi">10.48550/ARXIV.2511.21631</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ge</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Dang</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhong</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Wan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Ding</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Fu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Qwen2.5-VL technical report</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on February 19, 2025</comment>
          <pub-id pub-id-type="doi">10.48550/ARXIV.2502.13923</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Hui</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lv</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Ge</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Dang</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Bao</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Deng</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Xue</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Men</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Yin</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Fan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Su</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Cui</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Qiu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>Qwen3 technical report</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on May 14, 2025</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2505.09388</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Kwon</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>JW</given-names>
            </name>
          </person-group>
          <article-title>NestedFP: high-performance, memory-efficient dual-precision floating point support for LLMs</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on May 29, 2025</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2506.02024"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2506.02024</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kurtic</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Marques</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pandit</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kurtz</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Alistarh</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>"Give me BF16 or give me death"? Accuracy-performance trade-offs in LLM quantization</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on November 4, 2024</comment>
          <pub-id pub-id-type="doi">10.48550/ARXIV.2411.02355</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kwon</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhuang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sheng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>CH</given-names>
            </name>
            <name name-style="western">
              <surname>Gonzalez</surname>
              <given-names>JE</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Stoica</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Efficient memory management for large language model serving with PagedAttention</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on September 12, 2023</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2309.06180</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>WX</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Xia</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>Not all languages are created equal in LLMs: improving multilingual capability by cross-lingual-thought prompting</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on May 11, 2023</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2305.07004</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref40">
        <label>40</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ergen</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Logeswaran</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Jurgens</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Cross-lingual prompt steerability: towards accurate and robust LLM behavior across languages</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on December 2, 2025</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2512.02841</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref41">
        <label>41</label>
        <nlm-citation citation-type="web">
          <article-title>CIDER - clinical data extractor</article-title>
          <source>Bioinformatika LLM</source>
          <access-date>2026-08-07</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://llm.gyorffylab.com/cider">https://llm.gyorffylab.com/cider</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref42">
        <label>42</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Olaker</surname>
              <given-names>VR</given-names>
            </name>
            <name name-style="western">
              <surname>Fry</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Terebuh</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Davis</surname>
              <given-names>PB</given-names>
            </name>
            <name name-style="western">
              <surname>Tisch</surname>
              <given-names>DJ</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Miller</surname>
              <given-names>MG</given-names>
            </name>
            <name name-style="western">
              <surname>Dorney</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Palchuk</surname>
              <given-names>MB</given-names>
            </name>
            <name name-style="western">
              <surname>Kaelber</surname>
              <given-names>DC</given-names>
            </name>
          </person-group>
          <article-title>With big data comes big responsibility: strategies for utilizing aggregated, standardized, de-identified electronic health record data for research</article-title>
          <source>Clin Transl Sci</source>
          <year>2025</year>
          <month>01</month>
          <volume>18</volume>
          <issue>1</issue>
          <fpage>e70093</fpage>
          <pub-id pub-id-type="doi">10.1111/cts.70093</pub-id>
          <pub-id pub-id-type="medline">39740190</pub-id>
          <pub-id pub-id-type="pmcid">PMC11685181</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref43">
        <label>43</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Fareed</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Fatima</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Uddin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ahmed</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sattar</surname>
              <given-names>MA</given-names>
            </name>
          </person-group>
          <article-title>A systematic review of ethical considerations of large language models in healthcare and medicine</article-title>
          <source>Front Digit Health</source>
          <year>2025</year>
          <month>9</month>
          <day>11</day>
          <volume>7</volume>
          <fpage>1653631</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/fdgth.2025.1653631"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fdgth.2025.1653631</pub-id>
          <pub-id pub-id-type="medline">41019285</pub-id>
          <pub-id pub-id-type="pmcid">PMC12460403</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref44">
        <label>44</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Kuo</surname>
              <given-names>CF</given-names>
            </name>
          </person-group>
          <article-title>Roles and potential of Large language models in healthcare: a comprehensive review</article-title>
          <source>Biomed J</source>
          <year>2025</year>
          <month>10</month>
          <volume>48</volume>
          <issue>5</issue>
          <fpage>100868</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2319-4170(25)00042-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.bj.2025.100868</pub-id>
          <pub-id pub-id-type="medline">40311872</pub-id>
          <pub-id pub-id-type="pii">S2319-4170(25)00042-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC12517079</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref45">
        <label>45</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dennstädt</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hastings</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Putora</surname>
              <given-names>PM</given-names>
            </name>
            <name name-style="western">
              <surname>Schmerder</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Cihoric</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Implementing large language models in healthcare while balancing control, collaboration, costs and security</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>03</month>
          <day>06</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>143</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://boris-portal.unibe.ch/handle/20.500.12422/206555"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01476-7</pub-id>
          <pub-id pub-id-type="medline">40050366</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01476-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11885444</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref46">
        <label>46</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Meskó</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Topol</surname>
              <given-names>EJ</given-names>
            </name>
          </person-group>
          <article-title>The imperative for regulatory oversight of large language models (or generative AI) in healthcare</article-title>
          <source>NPJ Digit Med</source>
          <year>2023</year>
          <month>07</month>
          <day>06</day>
          <volume>6</volume>
          <issue>1</issue>
          <fpage>120</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-023-00873-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-023-00873-0</pub-id>
          <pub-id pub-id-type="medline">37414860</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-023-00873-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC10326069</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref47">
        <label>47</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sandmann</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hegselmann</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Fujarski</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Bickmann</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Wild</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Eils</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Varghese</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Benchmark evaluation of DeepSeek large language models in clinical decision-making</article-title>
          <source>Nat Med</source>
          <year>2025</year>
          <month>08</month>
          <volume>31</volume>
          <issue>8</issue>
          <fpage>2546</fpage>
          <lpage>9</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-025-03727-2</pub-id>
          <pub-id pub-id-type="medline">40267970</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-025-03727-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC12353792</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref48">
        <label>48</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tripathi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sukumaran</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Cook</surname>
              <given-names>TS</given-names>
            </name>
          </person-group>
          <article-title>Efficient healthcare with large language models: optimizing clinical workflow and enhancing patient care</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2024</year>
          <month>05</month>
          <day>20</day>
          <volume>31</volume>
          <issue>6</issue>
          <fpage>1436</fpage>
          <lpage>40</lpage>
          <pub-id pub-id-type="doi">10.1093/jamia/ocad258</pub-id>
          <pub-id pub-id-type="medline">38273739</pub-id>
          <pub-id pub-id-type="pii">7589687</pub-id>
          <pub-id pub-id-type="pmcid">PMC11105142</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref49">
        <label>49</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jonnagaddala</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>ZS</given-names>
            </name>
          </person-group>
          <article-title>Privacy preserving strategies for electronic health records in the era of large language models</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>01</month>
          <day>16</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>34</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01429-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01429-0</pub-id>
          <pub-id pub-id-type="medline">39820020</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01429-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC11739470</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref50">
        <label>50</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hendrycks</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Burns</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Basart</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mazeika</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Steinhardt</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Measuring massive multitask language understanding</article-title>
          <source>arXiv. Preprint posted online on September 7, 2020</source>
          <year>2026</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.2009.03300</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref51">
        <label>51</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Mihaylov</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Artetxe</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Simig</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Ott</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Goyal</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Bhosale</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Du</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Pasunuru</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Shleifer</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Koura</surname>
              <given-names>PS</given-names>
            </name>
            <name name-style="western">
              <surname>Chaudhary</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>O'Horo</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zettlemoyer</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Kozareva</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Diab</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Stoyanov</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Few-shot learning with multilingual language models</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on December 20, 2021</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2112.10668</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref52">
        <label>52</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Duan</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Yuan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>MMBench: is your multi-modal model an all-around player?</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on July 12, 2023</comment>
          <pub-id pub-id-type="doi">10.48550/ARXIV.2307.06281</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref53">
        <label>53</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Agrawal</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hegselmann</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Sontag</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <person-group person-group-type="editor">
            <name name-style="western">
              <surname>Goldberg</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Kozareva</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Large language models are few-shot clinical information extractors</article-title>
          <source>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</source>
          <year>2022</year>
          <publisher-loc>Stroudsburg, PA</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>1998</fpage>
          <lpage>2022</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref54">
        <label>54</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wornow</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Thapa</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Patel</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Steinberg</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Fleming</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pfeffer</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Fries</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NH</given-names>
            </name>
          </person-group>
          <article-title>The shaky foundations of clinical foundation models: a survey of large language models and foundation models for EMRs</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on March 22, 2023</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2303.12961</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref55">
        <label>55</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Santos</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Tariq</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Gichoya</surname>
              <given-names>JW</given-names>
            </name>
            <name name-style="western">
              <surname>Trivedi</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Banerjee</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Automatic classification of cancer pathology reports: a systematic review</article-title>
          <source>J Pathol Inform</source>
          <year>2022</year>
          <month>01</month>
          <day>20</day>
          <volume>13</volume>
          <fpage>100003</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2153-3539(22)00003-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jpi.2022.100003</pub-id>
          <pub-id pub-id-type="medline">35242443</pub-id>
          <pub-id pub-id-type="pii">S2153-3539(22)00003-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC8860734</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref56">
        <label>56</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hands</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Kavuluru</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>A survey of NLP methods for oncology in the past decade with a focus on cancer registry applications</article-title>
          <source>Artif Intell Rev</source>
          <year>2025</year>
          <volume>58</volume>
          <issue>10</issue>
          <fpage>314</fpage>
          <pub-id pub-id-type="doi">10.1007/s10462-025-11316-5</pub-id>
          <pub-id pub-id-type="medline">40688631</pub-id>
          <pub-id pub-id-type="pii">11316</pub-id>
          <pub-id pub-id-type="pmcid">PMC12267331</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref57">
        <label>57</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Vaid</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Menon</surname>
              <given-names>KM</given-names>
            </name>
            <name name-style="western">
              <surname>Freeman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Matteson</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Marin</surname>
              <given-names>ML</given-names>
            </name>
            <name name-style="western">
              <surname>Nadkarni</surname>
              <given-names>GN</given-names>
            </name>
          </person-group>
          <article-title>Using large language models to automate data extraction from surgical pathology reports: retrospective cohort study</article-title>
          <source>JMIR Form Res</source>
          <year>2025</year>
          <month>04</month>
          <day>07</day>
          <volume>9</volume>
          <fpage>e64544</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://formative.jmir.org/2025//e64544/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/64544</pub-id>
          <pub-id pub-id-type="medline">40194317</pub-id>
          <pub-id pub-id-type="pii">v9i1e64544</pub-id>
          <pub-id pub-id-type="pmcid">PMC11996145</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref58">
        <label>58</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Balasubramanian</surname>
              <given-names>JB</given-names>
            </name>
            <name name-style="western">
              <surname>Adams</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Roxanis</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>de Gonzalez</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Coulson</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Almeida</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>García-Closas</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Leveraging large language models for structured information extraction from pathology reports</article-title>
          <source>J Pathol Inform</source>
          <year>2025</year>
          <month>10</month>
          <day>10</day>
          <volume>19</volume>
          <fpage>100521</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2153-3539(25)00107-5"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jpi.2025.100521</pub-id>
          <pub-id pub-id-type="medline">41340635</pub-id>
          <pub-id pub-id-type="pii">S2153-3539(25)00107-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC12670944</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref59">
        <label>59</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jabal</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Warman</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Gupta</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Jain</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mazurowski</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wiggins</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Magudia</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Calabrese</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Language models and retrieval augmented generation for automated structured data extraction from diagnostic reports</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on September 15, 2024</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2409.10576</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref60">
        <label>60</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grothey</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Odenkirchen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Brkic</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schömig-Markiefka</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Quaas</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Büttner</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Tolkach</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Comprehensive testing of large language models for extraction of structured data in pathology</article-title>
          <source>Commun Med (Lond)</source>
          <year>2025</year>
          <month>03</month>
          <day>31</day>
          <volume>5</volume>
          <issue>1</issue>
          <fpage>96</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s43856-025-00808-8"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s43856-025-00808-8</pub-id>
          <pub-id pub-id-type="medline">40164789</pub-id>
          <pub-id pub-id-type="pii">10.1038/s43856-025-00808-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC11958830</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref61">
        <label>61</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bartels</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Carus</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>From text to data: open-source large language models in extracting cancer related medical attributes from German pathology reports</article-title>
          <source>Int J Med Inform</source>
          <year>2025</year>
          <month>11</month>
          <volume>203</volume>
          <fpage>106022</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1386-5056(25)00239-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.106022</pub-id>
          <pub-id pub-id-type="medline">40609461</pub-id>
          <pub-id pub-id-type="pii">S1386-5056(25)00239-4</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref62">
        <label>62</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shaitarova</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Zaghir</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lavelli</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Krauthammer</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rinaldi</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>Exploring the latest highlights in medical natural language processing across multiple languages: a survey</article-title>
          <source>Yearb Med Inform</source>
          <year>2023</year>
          <month>08</month>
          <volume>32</volume>
          <issue>1</issue>
          <fpage>230</fpage>
          <lpage>43</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://www.thieme-connect.com/DOI/DOI?10.1055/s-0043-1768726"/>
          </comment>
          <pub-id pub-id-type="doi">10.1055/s-0043-1768726</pub-id>
          <pub-id pub-id-type="medline">38147865</pub-id>
          <pub-id pub-id-type="pmcid">PMC10751112</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref63">
        <label>63</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Han</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Pechenizkiy</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Fang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>MuBench: assessment of multilingual capabilities of large language models across 61 languages</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on June 24, 2025</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2506.19468</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
