<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id>
      <journal-title>Journal of Medical Internet Research</journal-title>
      <issn pub-type="epub">1438-8871</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v28i1e86453</article-id>
      <article-id pub-id-type="pmid">42714033</article-id>
      <article-id pub-id-type="doi">10.2196/86453</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>“Small” Large Language Models in the Hospital: Evaluation Study on Real-World Data in a Resource-Constrained Setting</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Coristine</surname>
            <given-names>Andrew</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Zhou</surname>
            <given-names>Sicheng</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Chen</surname>
            <given-names>Frank</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Xu</surname>
            <given-names>He A</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-0248-8604</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Pythoud</surname>
            <given-names>Romain</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0003-9770-3043</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Thorball</surname>
            <given-names>Christian W</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-6869-6943</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Carra</surname>
            <given-names>Giorgia</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <xref rid="aff02" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-8002-224X</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Kulynych</surname>
            <given-names>Bogdan</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-5923-3931</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Despraz</surname>
            <given-names>Jérémie</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-2435-4079</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Galland-Decker</surname>
            <given-names>Coralie</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff03" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-8897-8473</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author">
          <name name-style="western">
            <surname>Maslias</surname>
            <given-names>Errikos</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff04" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1439-0276</ext-link>
        </contrib>
        <contrib id="contrib9" contrib-type="author">
          <name name-style="western">
            <surname>Baudson</surname>
            <given-names>Edouard</given-names>
          </name>
          <degrees>BSc</degrees>
          <xref rid="aff04" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0004-0723-3357</ext-link>
        </contrib>
        <contrib id="contrib10" contrib-type="author">
          <name name-style="western">
            <surname>Brahier</surname>
            <given-names>Thomas</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff02" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-9256-641X</ext-link>
        </contrib>
        <contrib id="contrib11" contrib-type="author">
          <name name-style="western">
            <surname>Kraege</surname>
            <given-names>Vanessa</given-names>
          </name>
          <degrees>MD, eMBA</degrees>
          <xref rid="aff05" ref-type="aff">5</xref>
          <xref rid="aff06" ref-type="aff">6</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-6654-8154</ext-link>
        </contrib>
        <contrib id="contrib12" contrib-type="author">
          <name name-style="western">
            <surname>de Sousa Teixeira</surname>
            <given-names>Ana Catarina</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff06" ref-type="aff">6</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0002-5810-6249</ext-link>
        </contrib>
        <contrib id="contrib13" contrib-type="author">
          <name name-style="western">
            <surname>Fidalgo</surname>
            <given-names>Carlos</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff07" ref-type="aff">7</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-1729-9821</ext-link>
        </contrib>
        <contrib id="contrib14" contrib-type="author">
          <name name-style="western">
            <surname>Berthaudin</surname>
            <given-names>Florian</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff07" ref-type="aff">7</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-2475-8527</ext-link>
        </contrib>
        <contrib id="contrib15" contrib-type="author">
          <name name-style="western">
            <surname>Madina</surname>
            <given-names>Amagoia</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff08" ref-type="aff">8</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0005-6187-374X</ext-link>
        </contrib>
        <contrib id="contrib16" contrib-type="author">
          <name name-style="western">
            <surname>Zoergiebel</surname>
            <given-names>Solange</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff08" ref-type="aff">8</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-5452-6704</ext-link>
        </contrib>
        <contrib id="contrib17" contrib-type="author">
          <name name-style="western">
            <surname>Bastardot</surname>
            <given-names>Francois</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff03" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-4060-0353</ext-link>
        </contrib>
        <contrib id="contrib18" contrib-type="author">
          <name name-style="western">
            <surname>Stravodimou</surname>
            <given-names>Athina</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff09" ref-type="aff">9</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-9608-985X</ext-link>
        </contrib>
        <contrib id="contrib19" contrib-type="author">
          <name name-style="western">
            <surname>Méan</surname>
            <given-names>Marie</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff07" ref-type="aff">7</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-0477-7899</ext-link>
        </contrib>
        <contrib id="contrib20" contrib-type="author">
          <name name-style="western">
            <surname>Fellay</surname>
            <given-names>Jacques</given-names>
          </name>
          <degrees>MD, PhD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <xref rid="aff10" ref-type="aff">10</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-8240-939X</ext-link>
        </contrib>
        <contrib id="contrib21" contrib-type="author">
          <name name-style="western">
            <surname>Ryvlin</surname>
            <given-names>Philippe</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff04" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-7775-6576</ext-link>
        </contrib>
        <contrib id="contrib22" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Raisaro</surname>
            <given-names>Jean Louis</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff01" ref-type="aff">1</xref>
          <address>
            <institution>Biomedical Data Science Center</institution>
            <institution>University Hospital of Lausanne</institution>
            <addr-line>Rue du Bugnon 21</addr-line>
            <addr-line>Lausanne, 1011</addr-line>
            <country>Switzerland</country>
            <phone>41 021 314 52 25</phone>
            <email>jean.raisaro@chuv.ch</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-2052-6133</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff01">
        <label>1</label>
        <institution>Biomedical Data Science Center</institution>
        <institution>University Hospital of Lausanne</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff02">
        <label>2</label>
        <institution>Infectious Diseases Service</institution>
        <institution>University Hospital of Lausanne</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff03">
        <label>3</label>
        <institution>Clinical Informatics Unit</institution>
        <institution>University Hospital of Lausanne</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff04">
        <label>4</label>
        <institution>Department of Clinical Neurosciences</institution>
        <institution>University Hospital of Lausanne</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff05">
        <label>5</label>
        <institution>Lausanne University hospital</institution>
        <institution>Innovation and medical research directorate</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff06">
        <label>6</label>
        <institution>University of Lausanne</institution>
        <institution>Faculty of Biology and Medicine</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff07">
        <label>7</label>
        <institution>Lausanne University Hospital and University of Lausanne</institution>
        <institution>Division of Internal Medicine</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff08">
        <label>8</label>
        <institution>IT Department</institution>
        <institution>University Hospital of Lausanne</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff09">
        <label>9</label>
        <institution>Department of Oncology, Medical Oncology Service</institution>
        <institution>University Hospital of Lausanne</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="aff10">
        <label>10</label>
        <institution>School of Life Sciences</institution>
        <institution>École Polytechnique Fédérale de Lausanne</institution>
        <addr-line>Lausanne</addr-line>
        <country>Switzerland</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Jean Louis Raisaro <email>jean.raisaro@chuv.ch</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>9</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>28</volume>
      <elocation-id>e86453</elocation-id>
      <history>
        <date date-type="received">
          <day>4</day>
          <month>11</month>
          <year>2025</year>
        </date>
        <date date-type="rev-request">
          <day>16</day>
          <month>1</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>5</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>5</day>
          <month>5</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©He A Xu, Romain Pythoud, Christian W Thorball, Giorgia Carra, Bogdan Kulynych, Jérémie Despraz, Coralie Galland-Decker, Errikos Maslias, Edouard Baudson, Thomas Brahier, Vanessa Kraege, Ana Catarina de Sousa Teixeira, Carlos Fidalgo, Florian Berthaudin, Amagoia Madina, Solange Zoergiebel, Francois Bastardot, Athina Stravodimou, Marie Méan, Jacques Fellay, Philippe Ryvlin, Jean Louis Raisaro. Originally published in the Journal of Medical Internet Research (https://www.jmir.org), 09.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on https://www.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.jmir.org/2026/1/e86453" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Large language models (LLMs) are increasingly being deployed in health care, but their use and deployment in many real-world hospital environments pose significant challenges and concerns. In particular, state-of-the-art commercial models store or process data externally, which is often in conflict with ensuring patient data protection. At the same time, using LLMs locally is limited by the lack of available computing infrastructure. Small open-source LLMs that do not require substantial computing resources could offer a practical way to resolve these tensions, but their medical utility in real-world local contexts, especially in non-English languages, has not been sufficiently evaluated.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to evaluate the feasibility of small, locally deployable open-source LLMs for clinically relevant tasks in a resource-constrained hospital setting and to propose a reproducible framework for institution-specific evaluation before deployment.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We evaluated 6 open-source LLMs ranging from 8B to 24B parameters (from the Mistral, Phi4, Falcon3, Llama3.1, and Meditron3 families) in a zero-shot setting across 7 tasks covering 4 clinical use cases: information extraction, medical text translation, text generation, and clinical decision support. We used deidentified French clinical data from a Swiss tertiary hospital, including discharge letters, clinical notes, and structured electronic health records. Performance was assessed using task-specific metrics, such as precision, recall, <italic>F</italic><sub>1</sub>-score, embedding-based semantic similarity, recall-oriented understudy for gisting evaluation (ROUGE) score, readability indices, and human review by clinicians.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Model performance varied substantially between tasks. In the simplest retrieval task (needle-in-the-haystack), several models performed strongly, with Llama3.1 achieving an <italic>F</italic><sub>1</sub>-score of 99.81% and Mistral-small achieving 99.71%. In contrast, performance was poor in more complex tasks. For detecting protected health information, the best-performing LLMs achieved only modest overall macro–<italic>F</italic><sub>1</sub>-scores (0.33-0.34), substantially below a fine-tuned Robustly Optimized BERT Pretraining Approach (RoBERTa) baseline (0.94). In the task of extracting immune-related adverse events from discharge notes, the highest overall macro–<italic>F</italic><sub>1</sub>-score was 0.35 with Phi4. For medical text translation, Phi4 ranked highest in embedding-based evaluation, whereas Meditron3-Phi4 performed the worst, with clinician reviews identifying hallucinations in 55% of its outputs. In the task of summarizing discharge letters, quality was low across all models, with the best penalized ROUGE score reaching only 0.169 with Llama3.1. In the tasks of generating patient-friendly discharge note summaries and clinical decision support, clinician ratings generally ranged from dissatisfied to neutral, and no model achieved consistently satisfactory performance.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>Small open-source LLMs appear feasible for simple retrieval-oriented tasks in local hospital deployments but are currently inadequate for more complex applications, such as clinical decision support, deidentification, extraction of adverse events, and medical summarization. These findings highlight the importance of locally grounded evaluation tailored to specific use cases and the need for robust institutional evaluation frameworks to ensure safe and reliable deployment.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>clinical NLP</kwd>
        <kwd>evaluation framework</kwd>
        <kwd>French medical text</kwd>
        <kwd>health care</kwd>
        <kwd>hospital deployment</kwd>
        <kwd>large language model</kwd>
        <kwd>LLM</kwd>
        <kwd>natural language processing</kwd>
        <kwd>resource-constrained settings</kwd>
        <kwd>small language models</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Large language models (LLMs) attain high performance in medical tasks such as exam-style question answering according to benchmark-based evaluations [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. As opposed to classical statistical or machine learning systems, which often require structured data, LLMs work directly with text, the predominant data representation in medicine and a convenient free-form interface for practitioners to interact with models [<xref ref-type="bibr" rid="ref3">3</xref>]. This versatility and the observed high performance in standardized benchmarks prompted advocates across academia [<xref ref-type="bibr" rid="ref4">4</xref>] and industry [<xref ref-type="bibr" rid="ref5">5</xref>] to call for broader adoption of LLMs into medical informatics systems, for example, for decision support or assistance in operational and administrative tasks [<xref ref-type="bibr" rid="ref6">6</xref>].</p>
      <p>Despite this recent push, the adoption of LLMs in hospital environments poses significant challenges and concerns. First, the utility of many current LLMs is constrained by the data on which they are trained and tested. A significant portion of medically specialized LLMs is pretrained or fine-tuned using publicly available datasets such as PubMed and tested on variations of MultiMedQA [<xref ref-type="bibr" rid="ref1">1</xref>] multiple-choice datasets. These data do not fully represent the complexity and heterogeneity of real-world clinical use cases and the nuances of clinical notes, limiting their generalizability and performance [<xref ref-type="bibr" rid="ref7">7</xref>]. Second, although many LLMs are pretrained on multilingual corpora, their performance in non-English clinical contexts remains insufficiently studied. In addition, the practical integration of LLMs into existing hospital information systems poses substantial technical, logistical, and ethical issues due to data protection mandates, high computational requirements, legacy interfaces and infrastructures, and the need to control the model life cycle [<xref ref-type="bibr" rid="ref8">8</xref>], all of which require careful consideration for safe and successful deployment [<xref ref-type="bibr" rid="ref9">9</xref>]. Moreover, the regulatory frameworks differ significantly across the world. For instance, the United States primarily regulates health AI through sectoral rules such as the Health Insurance Portability and Accountability Act (HIPAA) [<xref ref-type="bibr" rid="ref10">10</xref>] and US Food and Drug Administration (FDA) guidance, whereas in the European Union (EU), the primary regulatory frameworks are the General Data Protection Regulation (GDPR) and the forthcoming EU AI Act, both of which impose strict requirements on data processing, transparency, and model accountability. The regulations significantly restrict the sharing of health data with external cloud services and strongly motivate on-premises or locally hosted deployments to ensure compliance and maintain institutional control over patient information.</p>
      <p>Existing research has largely focused on evaluating LLMs (often exceeding 70 billion parameters) available through APIs as a service across various health care applications and on simulated or publicly available datasets [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. Yet, most health care institutions do not have the computational resources to locally host any of these models and cannot transfer sensitive patient data to the cloud due to ethical commitments, data protection regulations, and internal policies. In contrast, smaller LLMs (typically under 24 billion parameters), which are more suitable for resource-constrained settings and practical deployments, have received comparatively little attention.</p>
      <p>To address this gap, this study introduces a reproducible evaluation framework for assessing smaller, computationally efficient LLMs on institution-specific data. The objective is to go beyond standardized and open medical evaluation benchmarks that often fail to capture local specificities and language. We demonstrate the utility of this framework in 2 ways. First, we characterize the performance of these models under realistic operational conditions by evaluating a suite of smaller LLMs on authentic French-language clinical tasks within our hospital, a tertiary university hospital in Switzerland. Second, we provide a methodological template that enables researchers and other health care institutions to validate and select appropriate models for local deployment, particularly in environments with limited computational resources. This work ultimately investigates the viability of accessible LLMs to support clinical practice in local contexts.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Ethical Considerations</title>
        <p>The Legal Affairs Unit of Lausanne University Hospital confirmed that this study falls outside the scope of Swiss research legislation (Federal Act on Research involving Human Beings, Human Research Act, HRA, SR 810.30). Consequently, formal authorization from an ethics committee was not required. However, in accordance with institutional and legal standards, the study was conducted exclusively using data for which informed consent had been previously obtained.</p>
      </sec>
      <sec>
        <title>Overview</title>
        <p>In this section, we evaluated 4 representative use cases—information extraction, medical text translation, text generation, and clinical decision support—to reflect common scenarios encountered in hospital environments. Multiple language models were selected and assessed across these tasks to compare their performance.</p>
      </sec>
      <sec>
        <title>LLM Selection and Configuration</title>
        <p>Model selection was guided by institutional computational capacity and regulatory restrictions preventing the use of external APIs or cloud services for patient data. Accordingly, we limited our evaluation to open-source LLMs with 1 to 24 billion parameters and excluded larger models (eg, 70B), even in quantized form, to avoid confounding performance effects.</p>
        <p>Current approaches to leveraging open-source LLMs in health care typically involve two main strategies: (1) fine-tuning pretrained models on specific medical data to enhance their performance on specialized tasks or (2) improving either general-purpose or medically fine-tuned LLMs directly through prompt engineering or incorporating external data as part of prompts to guide the outputs for clinical relevance [<xref ref-type="bibr" rid="ref13">13</xref>]. We focused only on pretrained models because fine-tuning requires additional computational resources, which are usually not available in many health care institutions. Additionally, we selected models that were state of the art at the time of analysis.</p>
        <p>Given the model size constraint, we initially considered a diverse set of models within the 1 to 24 billion parameter range. A critical selection criterion was sufficient pretraining on French-language corpora, as the clinical data in our setting were predominantly in French. Furthermore, we required models to support a minimum context window of 8000 tokens to adequately process the typical length of medical documents encountered in our hospital. Models that did not meet these requirements were excluded. For instance, models such as BioGPT [<xref ref-type="bibr" rid="ref14">14</xref>] and BioMistral-7B [<xref ref-type="bibr" rid="ref15">15</xref>] were not considered due to their limited context window, while others (eg, MedLM [<xref ref-type="bibr" rid="ref16">16</xref>]) were excluded due to the lack of publicly available model weights, preventing reproducible evaluation. Similarly, BioMedGPT-LM-7B [<xref ref-type="bibr" rid="ref17">17</xref>] was excluded because, despite its domain-specific fine-tuning, it lacked instruction tuning necessary for the intended applications. Based on these criteria, we selected the following 6 open-source models for evaluation: Mistral-Small-24B-Instruct-2501 [<xref ref-type="bibr" rid="ref18">18</xref>], Phi4-14B [<xref ref-type="bibr" rid="ref19">19</xref>], Falcon3-10B-Instruct [<xref ref-type="bibr" rid="ref20">20</xref>], Llama3.1-8B-Instruct [<xref ref-type="bibr" rid="ref21">21</xref>], and Meditron3-Phi4-14B and Meditron3-8B-Instruct [<xref ref-type="bibr" rid="ref22">22</xref>]. While we acknowledge that several other high-performing models exist, limitations related to context window size and/or accessibility of model weights precluded their inclusion; future work may extend this evaluation to such models as they become available.</p>
        <p>To maintain consistency and assess the inherent capabilities of these smaller language models without the influence of few-shot examples—which can be challenging to optimize within limited context windows and may introduce variability—this study exclusively evaluated the LLMs in a zero-shot setting. This approach involves providing instructions directly to the model without any preceding examples or the help of external tools such as web search or retrieval-augmented generation (RAG) techniques [<xref ref-type="bibr" rid="ref23">23</xref>].</p>
        <p>All experiments were conducted on an internal server using the Hugging Face Transformers library (version 4.36.2). Hardware resources included 3 GPUs: 2 NVIDIA Quadro M6000 and 1 Quadro P6000, each with 24 GB of VRAM (72 GB VRAM in total). We evaluated dense transformer architectures without quantization. The pretrained model weights were obtained from the Hugging Face platform [<xref ref-type="bibr" rid="ref24">24</xref>]. For models natively distributed in Brain Floating Point 16-bit (BF16) precision, as our GPUs do not support this precision, the model weights were upcast to full 32-bit floating-point precision (FP32) during loading. This upcasting increased the memory footprint of the model weights. To facilitate inference within our 72 GB VRAM capacity, we used a hybrid offloading strategy using the <italic>device_map='auto'</italic> configuration. This sharded parts of the model layers across the 3 GPUs, while the remaining layers were offloaded to system RAM.</p>
        <p>Generation parameters were fixed for all use cases, employing greedy decoding with a maximum output sequence length of 8000 tokens. For reproducibility, prompt templates and system instructions are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
      </sec>
      <sec>
        <title>Data Acquisition and Preparation</title>
        <p>We conducted the study using deidentified patient data extracted from the clinical data warehouse of our hospital. We included only data from patients who had provided general informed consent for the secondary use of their clinical data for research purposes. The exact data used to test the use cases depended on the use case. Details for each use case are described in the following sections.</p>
        <p>We performed all data processing and analyses locally. To ensure patient confidentiality and comply with data protection regulations, all protected health information (PHI) was deidentified prior to its use in this research except for the named entity recognition (NER) use case (a subtask of the information extraction use case). We used a customized text deidentification pipeline previously developed and validated by a subset of the authors, as described in another study [<xref ref-type="bibr" rid="ref25">25</xref>]. To preserve the temporal integrity of medical information, all dates within each clinical document were shifted by the same randomly selected offset (between 1 and 3 years), thereby preserving relative temporal relationships and clinical timelines for clinically relevant use cases.</p>
        <p>To ensure the generation of clean, well-structured prompts for model input, we implemented a comprehensive data preprocessing pipeline. This process involved several key steps, as described in <xref rid="figure1" ref-type="fig">Figure 1</xref>. We presented all prompts in French and prompted the models to respond in French to ensure consistency with the language of the discharge letters, except where otherwise specified in the use-case descriptions.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Data extraction and preparation for use cases. Field (short text describing what type of information is presented in “content”) formats in the raw data were automatically detected, and empty or noninformative fields were removed. The remaining content was reordered based on clinician-defined preferences or task requirements, and irrelevant information was filtered out. Finally, the processed data were structured into standardized Markdown prompts, with titles and headers reworded to optimize clarity and relevance for the language models.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86453_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Use Case Selection and Design</title>
        <sec>
          <title>Information Extraction</title>
          <sec>
            <title>Needle-in-the-Haystack</title>
            <p>To assess the models’ capacity to locate specific information embedded within clinical documents, we implemented a needle-in-the-haystack evaluation designed to measure token-level information extraction rather than complex medical reasoning. The “needle” consisted of 2 interdependent fabricated sentences that were highly unlikely to occur naturally within our data: “The dinosaur is called Jackson” (“Le dinosaure s'appelle Jackson”) and “Jackson is purple” (“Jackson est pourpre”). These sentences, which collectively provide the answer to the question “What color is the dinosaur?” (“Quelle est la couleur du dinosaure?”), were strategically inserted at distinct points (the first and third quartiles) within each “haystack” document. In our implementation, both the needle and the question were in French. The fictional nature of these sentences ensured no semantic overlap with the original medical content, and their structure allowed for unambiguous answer verification, for example, through keyword matching. Responses provided in either English or French conveying the correct color (eg, “purple” or “pourpre”) were considered correct. We chose this approach to evaluate both the direct retrieval capabilities of the LLMs and their ability to connect related pieces of information separated by substantial intervening text.</p>
            <p>The haystack documents consisted of a dataset of 500 French medical discharge letters selected randomly from our data warehouse. We injected the needles into half of the documents and left the other half without needles. Each letter contained between 1000 and 3000 words and featured complex medical language, thereby constituting a challenging and clinically relevant test bed for LLM performance.</p>
          </sec>
          <sec>
            <title>NER</title>
            <p>Building on the assessment of information extraction, the subsequent task focused on NER to specifically extract PHI in clinical notes. We targeted 8 distinct PHI entities: “Name,” “Age,” “Address,” “Telephone,” “Date,” “Time,” and “ID,” selected from a subset of the HIPAA guidance. This evaluation aimed to assess the LLMs’ effectiveness in identifying sensitive personal data embedded within clinical narratives, a foundational capability for developing automated deidentification pipelines to limit privacy risks when sharing medical documentation across institutions or departments. We used an annotated NER dataset of 3687 records, including discharge letters, consultation letters, laboratory reports, and other clinical documents [<xref ref-type="bibr" rid="ref25">25</xref>], to test the performance of the LLMs.</p>
          </sec>
          <sec>
            <title>Immune-Related Adverse Event Detection</title>
            <p>To assess the LLMs’ performance in a realistic clinical context, the third task focused on the extraction of immune-related adverse events (irAEs) from discharge notes originating from our hospital’s oncology department. The targeted irAEs comprised colitis, thyroiditis, pneumonitis, skin-related effects, hepatitis, hypophysitis, rheumatological manifestations, cardiac toxicities, neurological complications, pancreatitis, nephrological issues, and diabetes. This assessment aimed to test the LLMs’ ability to accurately identify and retrieve specific, clinically significant information from complex medical narratives, mirroring a common requirement in routine oncological practice. We used an annotated irAE dataset of 400 discharge letters to test the performance of the LLMs.</p>
          </sec>
        </sec>
        <sec>
          <title>Medical Text Translation</title>
          <p>As demographics shift toward greater linguistic diversity, the translation of discharge notes into patients’ native languages has become a clinical and administrative imperative. This practice is necessary not only to bridge communication gaps and maintain continuity of care but also to ensure institutional compliance with legal mandates and accurate billing procedures [<xref ref-type="bibr" rid="ref26">26</xref>]. However, the automated translation of these documents presents a significant challenge. This difficulty is primarily attributed to the requirement for precise and contextually accurate translation of specialized medical terminology.</p>
          <p>In this use case, we focused on the models’ ability to accurately translate clinical discharge letters from French to English using a randomly selected dataset of 20 discharge letters collected from different clinical services.</p>
          <p>The choice of English as the target language was motivated by its status as a high-resource language in the training of LLMs and its importance for international clinical communication and research dissemination. Successful performance in this task therefore reflects not only translation capability but also accurate comprehension of specialized medical terminology. The prompt instructions were provided in English, and models were required to generate outputs in English (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This choice was motivated by technical considerations: most instruction-tuned LLMs are predominantly optimized for English instruction following, leading to more stable and reproducible behavior when prompts are issued in English [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. Using English prompts therefore allowed us to standardize evaluation conditions and reduce variability due to instruction misinterpretation. At the same time, this setup reflects a realistic clinical scenario in which non-English medical documents are translated into English for broader use. We acknowledge that this cross-lingual prompting strategy may introduce confounding effects, which are further discussed in the Limitations section.</p>
        </sec>
        <sec>
          <title>Text Generation</title>
          <sec>
            <title>Overview</title>
            <p>We assessed the clinical utility of LLMs using 2 text generation tasks targeting different end users. The first task assessed the generation of concise summaries of discharge letters intended for health care professionals. The second task evaluated the models’ ability to translate complex medical information into simplified, patient-friendly language.</p>
          </sec>
          <sec>
            <title>Generate Discharge Letter Summaries</title>
            <p>The first task tested how well the LLMs could summarize key events, diagnoses, and treatments from patient health records, reflecting the real-world challenge clinicians face in managing vast amounts of patient information and medical research. For this evaluation, we randomly selected 40 discharge letters from diverse clinical services and prompted the LLMs to generate a summary of approximately 300 words.</p>
          </sec>
          <sec>
            <title>Generate Patient-Friendly Discharge Notes</title>
            <p>The second task assessed the ability of LLMs to generate patient-friendly discharge notes. The main goal was to translate key clinical information—including diagnoses, treatments, and follow-up instructions—into language understandable to patients with diverse educational levels. There is evidence that this functionality can enhance patient-clinician communication, patient engagement, and continuity of care. For the experiment, we randomly selected 14 deidentified discharge letters from the internal medicine department. Although the prompt was designed in English for multilingual use, the model outputs were generated in French and were subsequently evaluated for clinical accuracy and clarity by 2 native French-speaking clinicians.</p>
          </sec>
        </sec>
        <sec>
          <title>Clinical Decision Support</title>
          <p>This use case assessed the LLMs’ capabilities in clinical decision support, specifically their ability to generate differential diagnoses and propose treatment plans. We consider this task to be one of the most challenging assessments of LLM performance in the medical domain.</p>
          <p>For this task, we intentionally excluded discharge letters and instead provided the LLMs with curated prompts containing comprehensive patient information, including anamnesis, consultation notes, laboratory results, vital signs, and results of clinical examinations. This design aimed to simulate a real-world clinical decision-making scenario in which the diagnosis had not yet been formally established or documented. The objective was to evaluate the models’ ability to integrate heterogeneous clinical data to generate plausible differential diagnoses and suggest appropriate management strategies. We randomly selected 10 deidentified medical cases from the neurology service for this task.</p>
        </sec>
      </sec>
      <sec>
        <title>Evaluation Methods</title>
        <sec>
          <title>Overview</title>
          <p>We evaluated model performance according to the specific nature of each use case, using metrics appropriate to the task. For tasks for which ground truth was available, such as information extraction or translation, we applied quantitative measures (eg, <italic>F</italic><sub>1</sub>-score and recall-oriented understudy for gisting evaluation [ROUGE] score [<xref ref-type="bibr" rid="ref29">29</xref>]). For tasks with more subjective or generative outputs, such as text generation or clinical decision support, we assessed performance through structured expert review. This approach ensures that comparisons reflect both task-specific accuracy and practical utility in a clinical context.</p>
        </sec>
        <sec>
          <title>Evaluation of Information Extraction</title>
          <p>We measured the performance of the LLMs on the information extraction tasks using ground truth data.</p>
          <p>For the needle-in-the-haystack subtask, we calculated recall, precision, and the <italic>F</italic><sub>1</sub>-score. Recall reflects the proportion of correct answers regarding the needle (dinosaur’s color). Precision reflects the proportion of retrieved answers that were correct, indicating the reliability of the model’s responses. The <italic>F</italic><sub>1</sub>-score, defined as the harmonic mean of precision and recall, provides a single summary measure that balances detection sensitivity and prediction accuracy.</p>
          <p>In the NER subtask for PHI, we assessed performance using precision, recall, and the macro-average <italic>F</italic><sub>1</sub>-score. We computed these metrics for each targeted PHI category by comparing LLM-extracted entities against a manually annotated ground truth dataset.</p>
          <p>Similarly, for the irAE extraction subtask, we compared the LLM outputs against expert-annotated immune-related adverse events in each discharge letter. We calculated precision, recall, and the macro-average <italic>F</italic><sub>1</sub>-score for each predefined irAE type to evaluate extraction performance.</p>
          <p>To evaluate the extraction performance of the LLMs, we used a previously fine-tuned Robustly Optimized BERT Pretraining Approach (RoBERTa)–based NER model as a baseline. The model was trained on 3010 annotated French clinical documents extracted from the Lausanne University Hospital data warehouse [<xref ref-type="bibr" rid="ref25">25</xref>]. Because the corpus is in French, we selected the CamemBERT model [<xref ref-type="bibr" rid="ref30">30</xref>] as the base model. The model was fine-tuned for 60 epochs using a learning rate of 2e–5 and a weight decay of 0.001. We used a consistent batch size of 32 for both training and evaluation phases. To optimize model selection, performance was evaluated at the end of each epoch, and the best-performing weights were automatically restored at the conclusion of the training process to ensure maximum generalizability.</p>
        </sec>
        <sec>
          <title>Evaluation of Medical Text Translation</title>
          <p>We used 2 approaches to evaluate the quality of medical text translation.</p>
          <p>First, we used an automated and scalable assessment of translation quality using an LLM-based embedding approach. We encoded both the original French source texts and their LLM-generated English translations into vector representations within a shared multilingual embedding space. We then quantified the semantic similarity between a source text and its translation by calculating the cosine similarity score between their respective vector embeddings. This approach is predicated on the principle that texts conveying the same meaning should exhibit high semantic overlap and thus be close in the embedding space, even when expressed in different languages [<xref ref-type="bibr" rid="ref31">31</xref>].</p>
          <p>For a robust analysis, we selected several distinct embedding models. The criteria for their inclusion were: (1) the model should have independent training processes to reduce shared biases, (2) it should be trained on multilingual data that included both French and English, (3) it should have a context window capacity of at least 8192 tokens to handle potentially long medical documents, and (4) it should have demonstrated strong overall performance on established embedding benchmarks. Thus, we chose the following models for this evaluation: gte-multilingual-base [<xref ref-type="bibr" rid="ref32">32</xref>], bge-m3 [<xref ref-type="bibr" rid="ref33">33</xref>], and nomic-embed-text-v1.5 [<xref ref-type="bibr" rid="ref34">34</xref>].</p>
          <p>We also conducted human evaluations by inviting 4 clinicians to evaluate 20 discharge letters randomly selected from various clinical services. They evaluated the quality of the translated text against the original French text using a 5-point Likert scale (1=very dissatisfied; 5=very satisfied). These criteria included:</p>
          <list list-type="bullet">
            <list-item>
              <p>Accuracy in translation: is the overall meaning preserved well?</p>
            </list-item>
            <list-item>
              <p>Completeness: are there any facts missing?</p>
            </list-item>
            <list-item>
              <p>Readability: does the translation sound natural and easy to read?</p>
            </list-item>
            <list-item>
              <p>Terminology: is the specialized vocabulary (especially medical terms) translated correctly?</p>
            </list-item>
            <list-item>
              <p>Safety in translation: does the translation contain any harmful mistranslation that may not be safe for patients? If there is harmful information, the translation is not considered safe.</p>
            </list-item>
            <list-item>
              <p>Structural quality: is the structure of the letter correctly preserved? Is there any misplaced information?</p>
            </list-item>
          </list>
          <p>We also asked the clinicians to indicate whether there were any hallucinations in the translated text (ie, whether the translation contained information that did not exist in the original letter). Clinicians assigned a score of 0 (no hallucination) or 1 (with hallucinations).</p>
          <p>To assess the consistency of rater scores across models and evaluation criteria, we computed a distribution-based agreement index for each model-criterion combination. Because the 4 clinicians evaluated nonoverlapping subsets of selected letters, direct pairwise comparison of individual scores was not possible. Instead, for each cell defined by a model m and criterion c, we first computed each clinician’s mean score across their assigned letters, yielding one aggregate score per clinician.</p>
          <p>As the original ratings were provided on a 5-point Likert scale (range 1-5), we linearly normalized them to the (0,1) interval prior to the analysis. We then quantified the dispersion of these rater-level means using the SD. Agreement was defined as:</p>
          <graphic xlink:href="jmir_v28i1e86453_fig7.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          <p>where σ<italic><sub>m,c</sub></italic> is the SD of the 4 clinicians’ normalized mean scores for model m and criterion c. Further, σ<italic><sub>max</sub></italic> (0.5) is the theoretical maximum SD for a bounded scale in the interval (0,1). This normalization maps agreement to a value between 0 and 1. A value of 0 indicates maximal disagreement, whereas a value of 1 indicates perfect consensus.</p>
        </sec>
        <sec>
          <title>Evaluation of Text Generation</title>
          <p>Evaluating the quality of LLM-generated medical summaries for clinicians presents distinct challenges. Although human assessment by medical professionals would offer the most direct and clinically relevant measure of an LLM’s summarization capabilities, aligning with their specific preferences and informational priorities, this approach was infeasible for the current study due to its time-intensive nature and the limited availability of expert evaluators. Consequently, we used automated benchmarking metrics. The cosine similarity approach, effective for the prior translation task, is less suitable for summarization due to several key limitations. First, significant structural differences exist between lengthy source documents and concise summaries (which often omit details such as extensive laboratory tables and may use formats such as bullet points). Second, maintaining semantic integrity in medical text requires the exact retention of domain-specific terms. Paraphrasing critical medical concepts can introduce ambiguity or error, which may not be reflected by embedding-based similarity measures. Finally, cosine similarity lacks a mechanism to penalize excessive length or direct copying. To address these issues, we used the penalized ROUGE score [<xref ref-type="bibr" rid="ref29">29</xref>] (equations 2-4) as a more suitable metric for evaluating these summaries.</p>
          <graphic xlink:href="jmir_v28i1e86453_fig8.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          <graphic xlink:href="jmir_v28i1e86453_fig9.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          <graphic xlink:href="jmir_v28i1e86453_fig10.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          <p>where <italic>Count<sub>match</sub></italic>(<italic>n</italic>-<italic>gram</italic>) represents the number of n-grams co-occurring in both the generated and reference texts, and <italic>Count(n-gram)</italic> is the total number of n-grams in the reference text. ROUGE-N scores range from 0 to 1, with higher values indicating better recall of key information. <italic>L<sub>c</sub></italic> represents the number of words in the generated text.</p>
          <p>Evaluation of patient-friendly discharge notes combined automated readability assessment with in-depth manual clinical review. We first applied an automated readability metric, the Kandel-Moles index [<xref ref-type="bibr" rid="ref35">35</xref>], to evaluate readability in French across a diverse set of discharge letters selected to cover various departments and specialties. This index is roughly analogous to the Flesch Reading Ease score for English. Then, the clinicians conducted manual assessments to verify the retention of key clinical facts and overall quality. This was conducted using a structured evaluation grid with 11 criteria, each rated on a 5-point Likert scale (1=very dissatisfied; 5=very satisfied). These criteria included:</p>
          <list list-type="bullet">
            <list-item>
              <p>Accuracy (diagnosis, treatment plan, patient instructions, and avoidance of distortions or hallucinations)</p>
            </list-item>
            <list-item>
              <p>Completeness (no omitted information)</p>
            </list-item>
            <list-item>
              <p>Safety (no clinically significant errors)</p>
            </list-item>
            <list-item>
              <p>Readability (ease of reading and minimal jargon)</p>
            </list-item>
            <list-item>
              <p>Tone (appropriate for patients)</p>
            </list-item>
            <list-item>
              <p>Logical structure (clear information hierarchy)</p>
            </list-item>
            <list-item>
              <p>Anonymization (no personal identifiers)</p>
            </list-item>
            <list-item>
              <p>Ethics (no biases)</p>
            </list-item>
            <list-item>
              <p>Overall report quality.</p>
            </list-item>
          </list>
          <p>Two clinicians independently evaluated 6 model-generated responses for each of 14 discharge letters to ensure a robust assessment.</p>
        </sec>
        <sec>
          <title>Evaluation of Clinical Decision Support</title>
          <p>Our initial attempts to automate the evaluation of LLM-generated clinical decision support proved ineffective. Standard automated metrics were unsuitable for 2 principal reasons. First, there was a structural mismatch. The LLM outputs, generated in a zero-shot setting, did not conform to the structure or format of clinician-authored notes. This rendered metrics based on semantic vector similarity, such as cosine similarity, unreliable for comparing the model’s response to the ground truth text. Second, there was a lexical discrepancy. Reference clinician reports were typically highly concise and contained numerous hospital-specific abbreviations and medical shorthand. Consequently, string-matching algorithms and metrics based on n-grams (eg, ROUGE) failed to capture semantic equivalence due to low lexical overlap. These challenges necessitated the development of a robust manual evaluation framework leveraging clinical expertise.</p>
          <p>We thus designed a manual evaluation protocol in collaboration with practicing clinicians. We recruited a panel of 3 board-certified clinicians to assess model outputs independently. They used a dedicated annotation platform to independently score the responses from 6 different LLMs across 10 unique clinical case prompts. Clinicians rated each response against 7 predefined criteria using a 5-point Likert scale (1=very dissatisfied; 5=very satisfied). The evaluation criteria were:</p>
          <list list-type="bullet">
            <list-item>
              <p>Diagnostic and treatment accuracy: correctness of the proposed diagnoses and treatment plans.</p>
            </list-item>
            <list-item>
              <p>Diagnostic and treatment safety: absence of suggestions that could lead to patient harm.</p>
            </list-item>
            <list-item>
              <p>Relevance of suggested additional examinations: clinical appropriateness of recommended diagnostic tests to confirm or refute differential diagnoses.</p>
            </list-item>
            <list-item>
              <p>Logical reasoning: coherence and logical consistency of the clinical reasoning presented.</p>
            </list-item>
            <list-item>
              <p>Structural quality: clarity, organization, and readability of the response.</p>
            </list-item>
          </list>
        </sec>
        <sec>
          <title>Statistical Test</title>
          <p>To test differences in model performance for the clinical decision support and patient-friendly discharge note generation use cases, we used the Kruskal-Wallis H test for each of the evaluation criteria defined above. Following significant Kruskal-Wallis test results, post hoc pairwise comparisons were performed using the Mann-Whitney <italic>U</italic> test. To control for type I errors across multiple comparisons, <italic>P</italic> values were adjusted using the Bonferroni correction. CIs were estimated using the bootstrap method with 1000 resamples.</p>
        </sec>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Overview</title>
        <p>For conciseness, in the following text we sometimes refer to the Mistral-Small-24B-Instruct-2501 model as “Mistral-small,” the Phi4-14B model as “Phi4,” the Llama3.1-8B-Instruct model as “Llama3.1,” the Falcon3-10B-Instruct model as “Falcon3,” the Meditron3-8B-Instruct model as “Meditron3,” and the Meditron3-Phi4-14B model as “Meditron3-Phi4.”</p>
      </sec>
      <sec>
        <title>Information Extraction</title>
        <p>We evaluated the 6 selected models across 3 information extraction tasks. The detailed performance metrics are presented in <xref ref-type="table" rid="table1">Tables 1</xref>-<xref ref-type="table" rid="table3">3</xref>, respectively. In the needle-in-the-haystack task, Llama3.1 achieved the highest retrieval performance (<italic>F</italic><sub>1</sub>-score=99.81%), with Mistral-small demonstrating comparable results. In contrast, Meditron3-Phi4 showed the weakest performance, exhibiting the lowest recall (61.35%) and an <italic>F</italic><sub>1</sub>-score of 76.15%.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Performance of the selected models on the needle-in-the-haystack task, measured by extraction recall, precision, F1-score, and 95% CI.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="360"/>
            <col width="110"/>
            <col width="110"/>
            <col width="130"/>
            <col width="290"/>
            <thead>
              <tr valign="top">
                <td>Model</td>
                <td>Size</td>
                <td>Recall, %</td>
                <td>Precision, %</td>
                <td><italic>F</italic><sub>1</sub>-score (95% CI), %</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Mistral-small</td>
                <td>24B</td>
                <td>99.42</td>
                <td>100</td>
                <td>99.71 (98.34-100)</td>
              </tr>
              <tr valign="top">
                <td>Phi4</td>
                <td>14B</td>
                <td>86.54</td>
                <td>100</td>
                <td>92.78 (88.23-96.13)</td>
              </tr>
              <tr valign="top">
                <td>Meditron3-Phi4</td>
                <td>14B</td>
                <td>61.35</td>
                <td>100</td>
                <td>76.05 (67.97-82.58)</td>
              </tr>
              <tr valign="top">
                <td>Falcon3</td>
                <td>10B</td>
                <td>95.58</td>
                <td>100</td>
                <td>97.74 (94.88-99.47)</td>
              </tr>
              <tr valign="top">
                <td>Llama3.1</td>
                <td>8B</td>
                <td>99.62</td>
                <td>100</td>
                <td>99.81 (98.34-100)</td>
              </tr>
              <tr valign="top">
                <td>Meditron3</td>
                <td>8B</td>
                <td>87.69</td>
                <td>89.76</td>
                <td> 88.71 (82.53-92.38)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Performance of the selected models on the PHIa detection task, measured by macro-average F1-score with 95% CI. Higher values indicate better performance. The column headings NAME, DATE, TIME, ID, AGE, TEL, and ADDRESS represent the original PHI entity categories.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="190"/>
            <col width="70"/>
            <col width="80"/>
            <col width="80"/>
            <col width="80"/>
            <col width="70"/>
            <col width="80"/>
            <col width="70"/>
            <col width="110"/>
            <col width="170"/>
            <thead>
              <tr valign="top">
                <td>Model</td>
                <td>Size</td>
                <td>NAME</td>
                <td>DATE</td>
                <td>TIME</td>
                <td>ID</td>
                <td>AGE</td>
                <td>TEL</td>
                <td>ADDRESS</td>
                <td>Overall (95% CI)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Mistral-small</td>
                <td>24B</td>
                <td>0.84<sup>b</sup></td>
                <td>0.11</td>
                <td>0.83</td>
                <td>0.26</td>
                <td>0.53<sup>b</sup></td>
                <td>0.06</td>
                <td>0.08<sup>b</sup></td>
                <td>0.33 (0.32-0.35)</td>
              </tr>
              <tr valign="top">
                <td>Phi4</td>
                <td>14B</td>
                <td>0.00</td>
                <td>0.77</td>
                <td>0.88</td>
                <td>0.51</td>
                <td>0.53<sup>b</sup></td>
                <td>0.06</td>
                <td>0.01</td>
                <td>0.34 (0.32-0.36)</td>
              </tr>
              <tr valign="top">
                <td>Meditron3-Phi4</td>
                <td>14B</td>
                <td>0.00</td>
                <td>0.89</td>
                <td>0.90<sup>b</sup></td>
                <td>0.46<sup>b</sup></td>
                <td>0.43</td>
                <td>0.06</td>
                <td>0.01</td>
                <td>0.34 (0.32-0.36)</td>
              </tr>
              <tr valign="top">
                <td>Falcon3</td>
                <td>10B</td>
                <td>0.00</td>
                <td>0.90<sup>b</sup></td>
                <td>0.87</td>
                <td>0.31</td>
                <td>0.53</td>
                <td>0.06</td>
                <td>0.04</td>
                <td>0.34 (0.31-0.35)</td>
              </tr>
              <tr valign="top">
                <td>Llama3.1</td>
                <td>8B</td>
                <td>0.00</td>
                <td>0.49</td>
                <td>0.61</td>
                <td>0.16</td>
                <td>0.26</td>
                <td>0.01</td>
                <td>0.04</td>
                <td>0.20 (0.17-0.22)</td>
              </tr>
              <tr valign="top">
                <td>Meditron3</td>
                <td>8B</td>
                <td>0.00</td>
                <td>0.00</td>
                <td>0.00</td>
                <td>0.00</td>
                <td>0.00</td>
                <td>0.00</td>
                <td>0.00</td>
                <td>0.00 (0.00-0.00)</td>
              </tr>
              <tr valign="top">
                <td>Fine-tuned RoBERTa</td>
                <td>—</td>
                <td>0.95</td>
                <td>0.99</td>
                <td>1.00</td>
                <td>0.81</td>
                <td>0.74</td>
                <td>1.00</td>
                <td>0.79</td>
                <td>0.94 (0.91-0.96)</td>
              </tr>
              <tr valign="top">
                <td>Support, n</td>
                <td>—</td>
                <td>541</td>
                <td>1336</td>
                <td>314</td>
                <td>25</td>
                <td>138</td>
                <td>166</td>
                <td>571</td>
                <td>3091</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>PHI: protected health information.</p>
            </fn>
            <fn id="table2fn2">
              <p><sup>b</sup>Highest performance among the tested large language models for each PHI entity category.</p>
            </fn>
            <fn id="table2fn3">
              <p><sup>c</sup>Not applicable.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Performance of the selected models on immune-related adverse event extraction, measured by macro-average F1-score with 95% CI. Higher values indicate better performance.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="210"/>
            <col width="140"/>
            <col width="110"/>
            <col width="160"/>
            <col width="120"/>
            <col width="130"/>
            <col width="130"/>
            <thead>
              <tr valign="top">
                <td>Category</td>
                <td>Mistral-small</td>
                <td>Phi4</td>
                <td>Meditron3-Phi4</td>
                <td>Falcon3</td>
                <td>Llama3.1</td>
                <td>Meditron3</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Cardiac</td>
                <td>0.36</td>
                <td>0.33</td>
                <td>0.40</td>
                <td>0.31</td>
                <td>0.00</td>
                <td>0.40</td>
              </tr>
              <tr valign="top">
                <td>Colitis</td>
                <td>0.46</td>
                <td>0.54</td>
                <td>0.53</td>
                <td>0.44</td>
                <td>0.54</td>
                <td>0.45</td>
              </tr>
              <tr valign="top">
                <td>Diabetes</td>
                <td>0.25</td>
                <td>0.55</td>
                <td>0.17</td>
                <td>0.00</td>
                <td>0.00</td>
                <td>0.00</td>
              </tr>
              <tr valign="top">
                <td>Hepatitis</td>
                <td>0.38</td>
                <td>0.36</td>
                <td>0.33</td>
                <td>0.29</td>
                <td>0.18</td>
                <td>0.09</td>
              </tr>
              <tr valign="top">
                <td>Hypophysitis</td>
                <td>0.31</td>
                <td>0.32</td>
                <td>0.31</td>
                <td>0.12</td>
                <td>0.12</td>
                <td>0.17</td>
              </tr>
              <tr valign="top">
                <td>Nephrological</td>
                <td>0.08</td>
                <td>0.22</td>
                <td>0.35</td>
                <td>0.24</td>
                <td>0.21</td>
                <td>0.29</td>
              </tr>
              <tr valign="top">
                <td>Neurological</td>
                <td>0.08</td>
                <td>0.11</td>
                <td>0.11</td>
                <td>0.03</td>
                <td>0.06</td>
                <td>0.00</td>
              </tr>
              <tr valign="top">
                <td>Pancreatitis</td>
                <td>0.55</td>
                <td>0.67</td>
                <td>0.00</td>
                <td>0.57</td>
                <td>0.67</td>
                <td>0.40</td>
              </tr>
              <tr valign="top">
                <td>Pneumonitis</td>
                <td>0.46</td>
                <td>0.53</td>
                <td>0.43</td>
                <td>0.29</td>
                <td>0.22</td>
                <td>0.00</td>
              </tr>
              <tr valign="top">
                <td>Rheumatological</td>
                <td>0.07</td>
                <td>0.09</td>
                <td>0.08</td>
                <td>0.00</td>
                <td>0.00</td>
                <td>0.00</td>
              </tr>
              <tr valign="top">
                <td>Skin</td>
                <td>0.21</td>
                <td>0.20</td>
                <td>0.26</td>
                <td>0.18</td>
                <td>0.29</td>
                <td>0.18</td>
              </tr>
              <tr valign="top">
                <td>Thyroiditis</td>
                <td>0.30</td>
                <td>0.34</td>
                <td>0.29</td>
                <td>0.31</td>
                <td>0.24</td>
                <td>0.14</td>
              </tr>
              <tr valign="top">
                <td>Overall (95% CI)</td>
                <td>0.29 (0.20-0.35)</td>
                <td>0.35 (0.25-0.42)</td>
                <td>0.26 (0.19-0.32)</td>
                <td>0.23 (0.15-0.28)</td>
                <td>0.21 (0.12-0.26)</td>
                <td>0.17 (0.08-0.25)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>Performance on the PHI detection task varied across models and PHI categories. Mistral-small achieved the best results for NAME, AGE, and ADDRESS detection. Falcon3 performed best for DATE and AGE, while Phi4 was most effective for ID and AGE. Meditron3-Phi4 yielded the highest performance for TIME detection. Despite these differences, no single model demonstrated consistently high performance across all PHI categories.</p>
        <p>We further compared LLM performance with that of a fine-tuned RoBERTa baseline trained on a French clinical dataset from our institution [<xref ref-type="bibr" rid="ref25">25</xref>], providing a reference point for assessing the performance gap between zero-shot small LLMs and a task-specific fine-tuned model.</p>
        <p>Similar to the PHI detection task, model performance in the irAE detection task differed substantially across irAE categories. Models from the Meditron family exhibited superior performance in detecting cardiac adverse events, with Meditron3-Phi4 also achieving the best results for nephrological adverse events. The Mistral-small model was most effective for hepatitis detection. For colitis detection, both the Phi4 and Llama3.1 models demonstrated leading performance. The Phi4 model additionally excelled at detecting diabetes and hypophysitis. Overall, no single model demonstrated consistently strong performance across all evaluated irAE categories.</p>
      </sec>
      <sec>
        <title>Medical Text Translation</title>
        <p>Because the exact values of cosine similarity scores are model dependent and not directly comparable, we used a relative comparison rather than an absolute one. We evaluated the models based on their performance rankings. <xref ref-type="table" rid="table4">Table 4</xref> summarizes these findings, presenting the overall rank and the detailed ranking for each generative model as determined by the various embedding-based similarity scores. Among the models tested, Phi4 demonstrated the highest translation efficacy, whereas Meditron3-Phi4 yielded the poorest results. Our analysis of the model outputs revealed that the performance degradation of Meditron3-Phi4 is attributable to its fine-tuning process, during which the output was truncated to a 1000-token window despite an 8000-token input capacity, which severely compromised its long-context translation capabilities.</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Overall performance ranking of the selected large language models for the medical translation task from French to English, based on individual ranking results from 3 independent embedding models.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="260"/>
            <col width="90"/>
            <col width="250"/>
            <col width="100"/>
            <col width="200"/>
            <col width="100"/>
            <thead>
              <tr valign="top">
                <td>Model</td>
                <td>Size</td>
                <td colspan="3">Embedding similarity ranking</td>
                <td>Overall rank</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>gte-multilingual-base</td>
                <td>bge-m3</td>
                <td>nomic-embed-text-v1.5</td>
                <td>
                  <break/>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Phi4</td>
                <td>14B</td>
                <td>1</td>
                <td>1</td>
                <td>1</td>
                <td>1</td>
              </tr>
              <tr valign="top">
                <td>Falcon3</td>
                <td>10B</td>
                <td>2</td>
                <td>2</td>
                <td>3</td>
                <td>2</td>
              </tr>
              <tr valign="top">
                <td>Mistral-small</td>
                <td>24B</td>
                <td>3</td>
                <td>3</td>
                <td>4</td>
                <td>3</td>
              </tr>
              <tr valign="top">
                <td>Meditron3</td>
                <td>8B</td>
                <td>5</td>
                <td>4</td>
                <td>2</td>
                <td>4</td>
              </tr>
              <tr valign="top">
                <td>Llama3.1</td>
                <td>8B</td>
                <td>4</td>
                <td>5</td>
                <td>5</td>
                <td>5</td>
              </tr>
              <tr valign="top">
                <td>Meditron3-Phi4</td>
                <td>14B</td>
                <td>6</td>
                <td>6</td>
                <td>6</td>
                <td>6</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>The suboptimal performance of Meditron3-Phi4 in the medical translation task was also confirmed by clinician evaluations (<xref rid="figure2" ref-type="fig">Figure 2</xref>). This model consistently received “dissatisfied” ratings across all evaluation dimensions, representing statistically significant underperformance compared with the other models. In contrast, Falcon3, Llama3.1, and Mistral-small consistently achieved or exceeded the “satisfied” threshold across all criteria. The Meditron3 and Phi4 models occupied an intermediate tier, with performance ratings aligning with the “neutral” level.</p>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Clinicians’ evaluation of the text generated by large language models (LLMs) in 3 use cases: medical text translation, clinical decision support, and patient-friendly discharge note generation across the predefined evaluation dimensions. Each colored bar represents a tested LLM. Error bars represent the 95% CI. *P&#60;.05; **P&#60;.001. See <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86453_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Regarding hallucinations, most models demonstrated high fidelity to the source text. As shown in <xref ref-type="table" rid="table5">Table 5</xref>, Mistral-small achieved a perfect score, with no identified hallucinations (0/20). Llama3.1 and Falcon3 followed closely, each exhibiting hallucinations in only 5% of cases (1/20). In contrast, the Meditron3-Phi4 model emerged as a significant outlier, generating nonexistent information in 55% of the evaluated summaries (mean 0.61, SD 0.32), representing a substantial deficit in clinical factuality compared with its counterparts. <xref rid="figure3" ref-type="fig">Figure 3</xref> presents the distribution-based agreement for each model-criterion combination. Agreement varied across combinations. The readability of the Falcon3 model (rated between “satisfied” and “very satisfied”), the accuracy of translation of the Meditron3-Phi4 model (rated “dissatisfied”), and hallucination ratings for the Mistral-small model (rated “no hallucination”) reached the highest agreement score (score 1, indicating perfect consensus). Clinicians’ agreement on the readability of the Phi4 model was the lowest.</p>
        <table-wrap position="float" id="table5">
          <label>Table 5</label>
          <caption>
            <p>Evaluation of hallucination rates across the language models (n=20).</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="440"/>
            <col width="300"/>
            <col width="260"/>
            <thead>
              <tr valign="top">
                <td>Model</td>
                <td>Score, mean (SD)</td>
                <td>Hallucination count, n/N (%)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Mistral-small</td>
                <td>0.00 (0.00)</td>
                <td>0/20 (0)</td>
              </tr>
              <tr valign="top">
                <td>Llama3.1</td>
                <td>0.05 (0.22)</td>
                <td>1/20 (5)</td>
              </tr>
              <tr valign="top">
                <td>Falcon3</td>
                <td>0.05 (0.23)</td>
                <td>1/20 (5)</td>
              </tr>
              <tr valign="top">
                <td>Phi4</td>
                <td>0.11 (0.32)</td>
                <td>2/20 (10)</td>
              </tr>
              <tr valign="top">
                <td>Meditron3</td>
                <td>0.11 (0.32)</td>
                <td>2/20 (10)</td>
              </tr>
              <tr valign="top">
                <td>Meditron3-Phi4</td>
                <td>0.61 (0.32)</td>
                <td>11/20 (55)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Raters’ agreement for the medical text translation use case. Distribution-based agreement index across models and evaluation criteria. Values range from 0 (maximal disagreement) to 1 (perfect consensus). Higher values indicate greater consistency across clinicians. The violin plots present the distribution of the agreement index across all evaluation criteria for each language model.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86453_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Text Generation</title>
        <sec>
          <title>Discharge Summary Generation</title>
          <p>We detail the performance of selected LLMs on the summarization task, quantified by our proposed penalized ROUGE (Pen ROUGE) metric, in <xref ref-type="table" rid="table6">Table 6</xref>. The results show that Llama3.1 was the top-performing model, whereas Falcon3 was the least effective. Two key observations emerge from these results. First, the performance gap between the models was small, indicating only minor relative advantages. Second, and more critically, the universally low Pen ROUGE scores (between 0.1 and 0.2) suggest that the overall summarization quality was suboptimal across all tested models, highlighting the inherent difficulty of the task for small LLMs.</p>
          <table-wrap position="float" id="table6">
            <label>Table 6</label>
            <caption>
              <p>Mean ROUGEa and Pen ROUGEb scores of the selected large language models on the discharge summary generation task, with 95% CIs.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="130"/>
              <col width="80"/>
              <col width="170"/>
              <col width="220"/>
              <col width="200"/>
              <col width="200"/>
              <thead>
                <tr valign="top">
                  <td>Model</td>
                  <td>Size</td>
                  <td>ROUGE↑ (95% CI)</td>
                  <td>Text length, words (95% CI)</td>
                  <td>Length penalty (95% CI)</td>
                  <td>Pen ROUGE↑ (95% CI)</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td>Mistral-small</td>
                  <td>24B</td>
                  <td>0.184 (0.14-0.22)</td>
                  <td>391 (348.98-432.06)</td>
                  <td>0.833 (0.71-0.95)</td>
                  <td>0.156 (0.11-0.19)</td>
                </tr>
                <tr valign="top">
                  <td>Phi4</td>
                  <td>14B</td>
                  <td>0.151 (0.11-0.19)</td>
                  <td>368 (345.52-389.53)</td>
                  <td>0.927 (0.88-0.96)</td>
                  <td>0.141 (0.09-0.18)</td>
                </tr>
                <tr valign="top">
                  <td>Meditron3-Phi4</td>
                  <td>14B</td>
                  <td>0.190 (0.12-0.25)</td>
                  <td>445 (353.49-536.18)</td>
                  <td>0.727 (0.54-0.91)</td>
                  <td>0.114 (0.06-0.16)</td>
                </tr>
                <tr valign="top">
                  <td>Falcon3</td>
                  <td>10B</td>
                  <td>0.107 (0.08-0.13)</td>
                  <td>248 (220.657-276.08)</td>
                  <td>0.998 (0.99-1.00)</td>
                  <td>0.107 (0.08-0.13)</td>
                </tr>
                <tr valign="top">
                  <td>Llama3.1</td>
                  <td>8B</td>
                  <td>0.180 (0.14-0.22)</td>
                  <td>316 (275.98-356.53)</td>
                  <td>0.941 (0.89-0.99)</td>
                  <td>0.169 (0.13-0.21)</td>
                </tr>
                <tr valign="top">
                  <td>Meditron3</td>
                  <td>8B</td>
                  <td>0.287 (0.17- 0.40)</td>
                  <td>412 (279.78-545.16)</td>
                  <td>0.730 (0.55-0.90)</td>
                  <td>0.144 (0.09-0.19)</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table6fn1">
                <p><sup>a</sup>ROUGE: recall-oriented understudy for gisting evaluation.</p>
              </fn>
              <fn id="table6fn2">
                <p><sup>b</sup>Pen ROUGE: penalized ROUGE.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
        </sec>
        <sec>
          <title>Clinical Decision Support</title>
          <p>The evaluation of diagnosis and treatment generation revealed suboptimal performance across all models, with none achieving a “satisfied” rating on any evaluation dimension (<xref rid="figure2" ref-type="fig">Figure 2</xref>). Most models scored within the “dissatisfied” to “neutral” range. Intermodel statistical analysis showed no significant variation except for accuracy of treatment, for which a significant difference was observed between Mistral-small and Meditron3 (<italic>P</italic>=.03). Although the Meditron models consistently yielded lower scores, the margin of underperformance remained statistically nonsignificant.</p>
          <p>Interrater agreement for this use case was evaluated using both observed agreement and Gwet agreement coefficient, version 1 (AC1) [<xref ref-type="bibr" rid="ref36">36</xref>], to account for potential bias due to skewed rating distributions (<xref rid="figure4" ref-type="fig">Figure 4</xref>). Overall, the observed agreement was consistently high across models and criteria, ranging from 0.70 to 0.90, with perfect agreement (1.00) observed for hallucination evaluation in several models. Correspondingly, AC1 values indicated moderate to strong agreement for most task-specific dimensions, including accuracy, safety, and test relevance (AC1 values 0.60-0.89 across models), suggesting robust and consistent annotation for these clinically relevant criteria.</p>
          <fig id="figure4" position="float">
            <label>Figure 4</label>
            <caption>
              <p>Raters’ agreement for the clinical decision support use case. Interrater agreement was measured by observed agreement and Gwet agreement coefficient, version 1 (AC1). Each column presents the ratings for 1 model; the model name is presented at the bottom of the plot. The color of the heat map represents the observed agreement value. The value in each cell presents both the observed agreement and AC1 value in the format observed agreement/(AC1). The violin plots present the distribution of AC1 values across all evaluation criteria for each language model. Horizontal reference lines indicate standard interpretation thresholds (excellent agreement indicated by the blue dotted line at an AC1 value of 0.75). Wider violins indicate greater variability in AC1 values across evaluation criteria.</p>
            </caption>
            <graphic xlink:href="jmir_v28i1e86453_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          </fig>
          <p>Agreement for logical reasoning and overall quality was more variable, with some models exhibiting lower AC1 values despite moderate observed agreement, reflecting greater subjectivity in these dimensions. Across models, Meditron-based variants generally demonstrated higher interrater consistency, while other models showed more variability depending on the evaluation criterion.</p>
          <p>Overall, the concordance between observed agreement and AC1 in this use case suggests that interrater reliability was generally robust, with less pronounced effects of skewed rating distributions compared with other evaluation settings.</p>
        </sec>
        <sec>
          <title>Patient-Friendly Discharge Notes</title>
          <p>To assess the readability of the generated summaries in French, as before, we used the Kandel-Moles readability index. The original discharge letters were found to have a medium readability level (<xref rid="figure5" ref-type="fig">Figure 5</xref>). Contrary to expectations, no model achieved a superior readability score (indicated by higher values in <xref rid="figure5" ref-type="fig">Figure 5</xref>), and in some instances, the automated generation resulted in text with lower readability than the original.</p>
          <fig id="figure5" position="float">
            <label>Figure 5</label>
            <caption>
              <p>Kandel-Moles readability index of sampled discharge letters (higher values indicate greater readability). Each colored bar represents a tested model. The error bars represent the 95% CI.</p>
            </caption>
            <graphic xlink:href="jmir_v28i1e86453_fig5.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          </fig>
          <p>However, evaluations by clinicians painted a more critical picture (<xref rid="figure2" ref-type="fig">Figure 2</xref>). Clinicians rated the generated summaries as “dissatisfied” for readability and “very dissatisfied” for tone. While models performed adequately on the dimensions of diagnostic accuracy, completeness, safety, and privacy (rated “neutral” to “satisfied”), they fell short on treatment plan accuracy, patient instructions, logical structure, and overall report quality (rated below “neutral”). Notably, all models consistently achieved a “very satisfied” rating in the ethics dimension.</p>
          <p>We noticed that the Meditron3 family consistently underperformed relative to the other models (<xref rid="figure2" ref-type="fig">Figure 2</xref>). This finding is counterintuitive, as these medically fine-tuned models performed substantially worse than the general-purpose base models and Mistral-small. We hypothesize that this performance degradation stems from trade-offs during continual fine-tuning, whereby specialization for medical knowledge may have compromised foundational capabilities such as reasoning, multilingual comprehension, or instruction following. We leave further investigation of this phenomenon to future work.</p>
          <p>As before, interrater agreement was assessed using both observed agreement and Gwet AC1 (<xref rid="figure6" ref-type="fig">Figure 6</xref>). Overall, observed agreement was consistently high across most evaluation criteria and models, frequently exceeding 0.70 and reaching perfect agreement (1.00) for dimensions such as tone and ethics. In contrast, AC1 values showed greater variability, with high agreement observed for readability, tone, ethics, and satisfaction (AC1 values 0.73-1.00 across models), indicating strong and consistent rater alignment for these subjective quality dimensions.</p>
          <fig id="figure6" position="float">
            <label>Figure 6</label>
            <caption>
              <p>Raters’ agreement for the patient-friendly discharge note generation use case. Interrater agreement was measured by observed agreement and Gwet agreement coefficient, version 1 (AC1). Each column presents the ratings for 1 model; the model name is presented at the bottom of the plot. The color of the heat map represents the observed agreement value. The value in each cell presents both the observed agreement and AC1 value in the format observed agreement/(AC1). The violin plots present the distribution of AC1 values across all evaluation criteria for each language model. Horizontal reference lines indicate standard interpretation thresholds (excellent agreement indicated by the blue dotted line at an AC1 value of 0.75). Wider violins indicate greater variability in AC1 values across evaluation criteria.</p>
            </caption>
            <graphic xlink:href="jmir_v28i1e86453_fig6.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          </fig>
          <p>For more task-specific criteria, such as accuracy and completeness, agreement was moderate and model dependent, with higher consistency observed for certain models (eg, Meditron3-Phi4 and Phi4) compared with others. Criteria related to safety and anonymization showed low or even negative AC1 values across several models, despite moderate observed agreement. This discrepancy reflects the known sensitivity of chance-corrected metrics to skewed label distributions and suggests that these dimensions were both challenging to assess consistently and prone to imbalance in rating categories.</p>
          <p>Interrater agreement was generally higher and more consistent in the clinical decision support use case than in patient-friendly discharge note generation, as reflected by stronger concordance between observed agreement and AC1. In contrast, the patient-oriented use case exhibited greater variability and more frequent discrepancies between the 2 metrics, likely reflecting increased subjectivity and skewed rating distributions in evaluating language quality and communication aspects.</p>
        </sec>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>Our study evaluated the performance of small, resource-efficient LLMs across multiple clinically relevant tasks, including information extraction, medical text translation, discharge summary generation, and clinical decision support. Overall, although several models demonstrated competence on simple information extraction tasks (eg, needle-in-the-haystack), performance consistently deteriorated as task complexity increased. In particular, tasks requiring nuanced clinical understanding, complex instruction following, or domain-specific reasoning—such as PHI detection, irAE extraction, discharge summarization, and clinical decision support—had low accuracy and poor reliability.</p>
        <p>In the PHI detection task, no model demonstrated consistently high performance across all PHI categories, and several models repeatedly failed to identify entities. Qualitative inspection of model outputs suggests that this is a systematic limitation rather than an isolated error. For example, the Meditron3-8B-Instruct model consistently responded with “John Doe” as the detected name for all tested samples. This behavior may be an unintended consequence of safety-oriented posttraining fine-tuning [<xref ref-type="bibr" rid="ref37">37</xref>] or may reflect inherent constraints of the models for this task. In comparison with the fine-tuned RoBERTa baseline, zero-shot LLMs appear to lack the capacity to reliably detect or generalize across diverse PHI representations. We hypothesize that incorporating a few-shot learning strategy could substantially improve performance.</p>
        <p>In the medical text translation use case, our results indicate that while Phi4 excelled in automated metrics, Falcon3, Llama3.1, and Mistral-small consistently received “satisfied” scores in clinician evaluations. Meditron3-Phi4 failed to show efficacy in automated evaluation metrics and produced translations with a high hallucination rate.</p>
        <p>None of the evaluated models produced discharge notes that met the satisfaction criteria of either clinicians or patients. Although summaries generally adhered to ethical and safety constraints—likely reflecting safety posttraining—their clinical completeness, correctness, and usefulness were consistently inadequate.</p>
        <p>Similarly, performance on clinical decision support tasks was poor across all tested small LLMs. This contrasts sharply with prior reports of strong performance for larger models such as GPT-4 and domain-specialized systems like Med-PaLM 2 on diagnostic benchmarks. Collectively, these results indicate that the evaluated small LLMs are not yet suitable for deployment in clinical applications requiring deep medical understanding, multistep reasoning, or precise instruction adherence.</p>
        <p>Taken together, our findings directly address the study’s primary objective: assessing whether small, resource-efficient LLMs are ready for real-world clinical deployment. Despite their theoretical appeal, current models exhibit insufficient performance across most tested use cases, particularly those involving complex clinical cognition.</p>
      </sec>
      <sec>
        <title>Comparison With Prior Work</title>
        <p>For clinical documentation, specifically discharge summary generation, our study revealed that the evaluated small LLMs were unable to produce satisfactory summaries for either clinicians or patients. This aligns with the difficulties reported even for larger models. For instance, prior work using a fine-tuned Llama-3 model and commercial GPT-4 and GPT-3.5 models [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>] identified notable error rates, inaccuracies, and omissions in LLM-generated summaries for clinicians. Our results with small LLMs indicate an even more pronounced performance deficit, suggesting that the intricate task of synthesizing complex medical narratives into accurate and complete clinician-facing summaries likely requires model capacities beyond those of the small LLMs tested.</p>
        <p>The contrast is also stark when considering patient-friendly summaries. While studies by Ganzinger et al [<xref ref-type="bibr" rid="ref38">38</xref>] using GPT-4 and Williams et al [<xref ref-type="bibr" rid="ref39">39</xref>] with GPT-3.5 demonstrated LLMs’ ability to significantly improve readability and produce largely accurate simplified texts, our small LLMs did not achieve this. This disparity underscores that the architectural scale and training data extent of models like GPT-4 are likely critical for nuanced rephrasing and simplification tasks.</p>
        <p>It is possible that the issues we found are less prominent when using large-scale LLMs. Their practical implementation and deployment, however, face significant challenges. Foremost among these are concerns regarding output reliability. For instance, studies deploying even advanced models like GPT-4 for discharge summarization have reported hallucinations or critical detail omissions in approximately 40% of cases [<xref ref-type="bibr" rid="ref39">39</xref>]. This underscores broader safety and reliability issues, indicating that a substantial proportion of LLM applications in medicine exhibit inaccuracies, rendering them insufficiently reliable for autonomous clinical use without rigorous human oversight. Furthermore, the seamless integration of these models into existing clinical workflows remains largely conceptual, with most current applications being retrospective or prototypical. The “black-box” nature of many LLMs also poses challenges for explainability, a critical factor for trust and adoption in clinical decision-making.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>Our study has several limitations that should be considered when interpreting the results and planning for future work. First, to ensure consistency, all experiments were conducted in a zero-shot setting, which may limit the performance of the models. Few-shot prompting may improve performance, especially for information extraction tasks such as NER detection.</p>
        <p>Second, our model selection did not include the full and rapidly evolving landscape of small LLMs. Our analysis focused on dense transformer architectures. Consequently, these findings may not generalize to alternative architectures, such as Mixture-of-Experts, or to domain-specialized models (eg, the MedGemma family [<xref ref-type="bibr" rid="ref40">40</xref>]), which may offer different performance trade-offs.</p>
        <p>Third, we observed a systematic failure in the Meditron3-8B-Instruct model in the PHI detection task, characterized by the recurrent generation of placeholders (eg, John Doe) rather than identifying actual entities. We hypothesize that this behavior stems from aggressive safety alignment during posttraining, which may inadvertently interfere with the model’s utility for clinical deidentification tasks. However, as we lacked access to non–safety-tuned checkpoints for a direct comparative analysis, this remains a preliminary observation that requires further investigation. From a practical perspective, these findings highlight the need for mitigation strategies when deploying LLMs for PHI detection. In particular, task-specific supervised fine-tuning on annotated clinical corpora may help recover entity recognition performance. In addition, hybrid approaches that combine rule-based methods (eg, pattern matching or deterministic deidentification pipelines) with model-based extraction could improve robustness and reduce the risk of systematic failures.</p>
        <p>Furthermore, the structural heterogeneity of the discharge letters used in the medical translation use case may have influenced the observed performance. Clinician feedback highlighted that some of the sampled letters are from surgical services, which often feature simpler structures and fewer technical sections compared with those from other medical services. Consequently, the inclusion of these relatively simpler documents may have inflated the perceived efficacy of the models. The current evaluation did not stratify results by clinical service or document complexity, which potentially limits the generalizability of our findings to more linguistically dense or multifaceted medical discourse.</p>
        <p>Finally, the key limitation of this study is the relatively small sample size used for evaluation. However, the objective was not to perform an exhaustive comparison of models but to assess whether current systems meet a minimum threshold for clinical utility. To mitigate potential sampling bias, cases were randomly selected, and their representativeness was verified using Kolmogorov-Smirnov tests comparing key attributes (eg, visit count, stay duration, and number and length of documents) against the full dataset; no statistically significant differences were observed (all <italic>P</italic>&#62;.16). Details about the test can be found in the supplementary materials. Across these samples, all models exhibited recurrent and clinically significant errors, indicating that they are not yet suitable for deployment in safety-critical settings. From a safety auditing perspective, such failures observed even in a small sample (eg, N=10) are sufficient to demonstrate unreliability. While increasing the sample size would improve the precision of performance estimates, it is unlikely to change the overall qualitative conclusion. Future work will focus on expanding the evaluation cohort through continued data collection with clinician input, as well as systematically quantifying interrater agreement to further assess the robustness and reliability of these inherently subjective evaluations.</p>
      </sec>
      <sec>
        <title>Future Directions</title>
        <p>Based on the findings and limitations of our study, we propose several directions for future research.</p>
        <p>First, future work could transition from zero-shot evaluations to few-shot prompting, chain-of-thought prompting, and test-time scaling strategies. Such research could investigate how the inclusion of domain-specific examples influences model performance. This could help determine the limits of small LLMs’ capabilities in clinical settings.</p>
        <p>A dedicated comparative analysis between base models and their safety-tuned counterparts is necessary to confirm our hypothesis regarding PHI detection failures. This research could focus on the potential loss in clinical task performance due to safety fine-tuning and developing system-level prompts that can bypass unintended refusals while maintaining safety requirements.</p>
        <p>Another notable limitation of our methodology is the bilingual disparity between the prompt instructions (English) and the source material (French) for the medical text translation task. Such cross-lingual prompting can introduce confounding effects, as models are often more heavily optimized for instruction following in English than in other languages. To address this, future work should include a comparative analysis of prompt-language effects to determine whether the observed performance deficits are inherent to the translation task or artifacts of the instruction language.</p>
        <p>Furthermore, future research should incorporate more granular, stratified samples of clinical notes or letters across diverse clinical specialties. Specifically, there is a need to evaluate model performance on high-complexity letters, such as those from internal medicine or oncology, which often involve complex multimorbidities and nuanced therapeutic reasoning. Future studies should aim to quantify how document length, structural density, and the presence of complex reasoning chains affect performance across medical use cases.</p>
        <p>To improve the generalizability of our findings, future work should involve larger, multidisciplinary panels of clinicians from diverse medical specialties. Additionally, conducting patient-side comprehension studies is important to faithfully validate the quality of generated patient-friendly discharge notes.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>The substantial computational resources and stringent data governance requirements associated with large LLMs present considerable barriers for many healthcare institutions. A significant number of hospitals and clinics, particularly those with limited resources or those operating under strict data protection regulations (eg, GDPR and HIPAA), may be unable to deploy large proprietary models locally or use cloud-based APIs due to cost, infrastructure limitations, or data residency and security concerns. It was precisely these practical deployment constraints that motivated our investigation into the clinical utility of smaller, more resource-efficient LLMs.</p>
        <p>Our findings demonstrate a critical disconnect between the theoretical promise of smaller LLMs and their practical utility in a specific clinical environment. Although these models offer a feasible deployment pathway for resource-constrained settings, their performance on our local, French-language data was marked by significant failures in complex reasoning and instruction adherence. This discrepancy demonstrates precisely why standardized benchmarks are insufficient; they fail to account for the local specificities and linguistic nuances that dictate real-world efficacy.</p>
        <p>The primary contribution of this work is therefore not merely to report on model limitations but to provide health care institutions with a methodological tool, as an evaluation protocol, through which they can generate their own evaluation benchmarks, ensuring that any deployed AI is genuinely fit for its intended clinical purpose. Concretely, this protocol consists of four key steps: (1) task scoping to define use cases and error profiles; (2) local dataset construction using representative, real-world data; (3) multidimensional evaluation of performance, reasoning, and safety; and (4) iterative failure analysis to refine model deployment.</p>
        <p>By emphasizing locally grounded validation over generalized performance claims, this protocol enables institutions to move from passive adoption of AI systems to active, evidence-based qualification. In doing so, it establishes a practical pathway toward safer and more context-sensitive integration of language models into clinical workflows.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Full prompt templates and variable definitions used in zero-shot experiments.</p>
        <media xlink:href="jmir_v28i1e86453_app1.docx" xlink:title="DOCX File , 34 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Additional experimental results and extended tables.</p>
        <media xlink:href="jmir_v28i1e86453_app2.docx" xlink:title="DOCX File , 84 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">AC1</term>
          <def>
            <p>agreement coefficient, version 1</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">BF16</term>
          <def>
            <p>Brain Floating Point 16-bit</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">EU</term>
          <def>
            <p>European Union</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">FDA</term>
          <def>
            <p>US Food and Drug Administration</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">FP32</term>
          <def>
            <p>32-bit floating-point precision</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">GDPR</term>
          <def>
            <p>General Data Protection Regulation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">HIPAA</term>
          <def>
            <p>Health Insurance Portability and Accountability Act</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">irAE</term>
          <def>
            <p>immune-related adverse event</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">Pen ROUGE</term>
          <def>
            <p>penalized ROUGE</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb11">PHI</term>
          <def>
            <p>protected health information</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb12">RAG</term>
          <def>
            <p>retrieval-augmented generation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb13">RoBERTa</term>
          <def>
            <p>Robustly Optimized BERT Pretraining Approach</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb14">ROUGE</term>
          <def>
            <p>recall-oriented understudy for gisting evaluation</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors used Google’s Gemini 3 (Flash) to assist in refining the language and style of the manuscript to enhance clarity. All original ideas, data collection, statistical analyses, and original drafts were performed by the authors without the use of AI. No patient data or sensitive hospital information were shared with or processed by an external AI tool. The final manuscript was manually reviewed and approved by all authors.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>The datasets generated or analyzed during this study are not publicly available due to confidentiality and data protection restrictions associated with patient data from Lausanne University Hospital but may be available in de-identified or aggregated form from the corresponding author on reasonable request, subject to institutional approval and applicable data protection regulations.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>This study was funded by the Swiss National Science Foundation “10003518 – Medical, Multilingual and Privacy-Preserving Natural Language Processing in the Clinical Domain” project and the Swiss National Science Foundation “237378 - Bridging Regulatory Data Protection Standards and Model Sharing in Healthcare” project.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>HAX contributed to the conceptualization, methodology, formal analysis, data anonymization, visualization, and preparation of the original draft. RP contributed to the investigation and data curation, while BK contributed to methodology and manuscript review and editing. GC and JD contributed to manuscript review and editing. CWT, AS, JF, and PR contributed resources, including the annotated dataset, validation, and manuscript review and editing for the immunotherapy adverse event use case. CG-D, EM, EB, TB, and F Bastardot contributed resources, validation, evaluation, and manuscript review and editing for the clinical decision support and patient-friendly discharge notes use cases. VK, ACST, CF, F Berthaudin, and MM contributed resources, validation, evaluation, and manuscript review and editing for the medical text translation use case. AM and SZ contributed to data extraction. JLR contributed resources, supervision, project administration, funding acquisition, and manuscript review and editing. All authors have read and agreed to the published version of the manuscript.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tanwani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Seneviratne</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gamble</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Babiker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schärli</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhery</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tomasev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>180</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bedi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cui</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Fuentes</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Unell</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wornow</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Banda</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Kotecha</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Keyes</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mai</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Oez</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Qiu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Jain</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Schettini</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Kashyap</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Fries</surname>
              <given-names>JA</given-names>
            </name>
          </person-group>
          <article-title>Holistic evaluation of large language models for medical tasks with MedHELM</article-title>
          <source>Nat Med</source>
          <year>2026</year>
          <volume>32</volume>
          <issue>3</issue>
          <fpage>943</fpage>
          <lpage>951</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-025-04151-2</pub-id>
          <pub-id pub-id-type="medline">41559415</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-025-04151-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC13267972</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Qiu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lam</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Acharya</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>TY</given-names>
            </name>
            <name name-style="western">
              <surname>Darzi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Yuan</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Topol</surname>
              <given-names>EJ</given-names>
            </name>
          </person-group>
          <article-title>LLM-based agentic systems in medicine and healthcare</article-title>
          <source>Nat Mach Intell</source>
          <year>2024</year>
          <volume>6</volume>
          <issue>12</issue>
          <fpage>1418</fpage>
          <lpage>1420</lpage>
          <pub-id pub-id-type="doi">10.1038/s42256-024-00944-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Topol</surname>
              <given-names>EJ</given-names>
            </name>
          </person-group>
          <article-title>The rise of agentic AI teammates in medicine</article-title>
          <source>Lancet</source>
          <year>2025</year>
          <volume>405</volume>
          <issue>10477</issue>
          <fpage>457</fpage>
          <pub-id pub-id-type="doi">10.1016/S0140-6736(25)00202-8</pub-id>
          <pub-id pub-id-type="medline">39922663</pub-id>
          <pub-id pub-id-type="pii">S0140-6736(25)00202-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hao</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <source>Empire of AI: Dreams and Nightmares in Sam Altman's OpenAI</source>
          <year>2025</year>
          <publisher-loc>New York</publisher-loc>
          <publisher-name>Penguin Press</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Clusmann</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kolbinger</surname>
              <given-names>FR</given-names>
            </name>
            <name name-style="western">
              <surname>Muti</surname>
              <given-names>HS</given-names>
            </name>
            <name name-style="western">
              <surname>Carrero</surname>
              <given-names>ZI</given-names>
            </name>
            <name name-style="western">
              <surname>Eckardt</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Laleh</surname>
              <given-names>NG</given-names>
            </name>
            <name name-style="western">
              <surname>Löffler</surname>
              <given-names>Chiara Maria Lavinia</given-names>
            </name>
            <name name-style="western">
              <surname>Schwarzkopf</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Unger</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Veldhuizen</surname>
              <given-names>GP</given-names>
            </name>
            <name name-style="western">
              <surname>Wagner</surname>
              <given-names>SJ</given-names>
            </name>
            <name name-style="western">
              <surname>Kather</surname>
              <given-names>JN</given-names>
            </name>
          </person-group>
          <article-title>The future landscape of large language models in medicine</article-title>
          <source>Commun Med (Lond)</source>
          <year>2023</year>
          <month>10</month>
          <day>10</day>
          <volume>3</volume>
          <issue>1</issue>
          <fpage>141</fpage>
          <pub-id pub-id-type="doi">10.1038/s43856-023-00370-1</pub-id>
          <pub-id pub-id-type="medline">37816837</pub-id>
          <pub-id pub-id-type="pii">10.1038/s43856-023-00370-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC10564921</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Agrawal</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>IY</given-names>
            </name>
            <name name-style="western">
              <surname>Gulamali</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Joshi</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>The evaluation illusion of large language models in medicine</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>10</month>
          <day>07</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>600</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01963-x</pub-id>
          <pub-id pub-id-type="medline">41057566</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01963-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC12504413</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Carra</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Kulynych</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Bastardot</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Kaufmann</surname>
              <given-names>DE</given-names>
            </name>
            <name name-style="western">
              <surname>Boillat-Blanco</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Raisaro</surname>
              <given-names>JL</given-names>
            </name>
          </person-group>
          <article-title>Participatory assessment of large language model applications in an academic medical center</article-title>
          <source>arXiv. Preprint posted online on December 9, 2024</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2501.10366"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2501.10366</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kwan</surname>
              <given-names>HY</given-names>
            </name>
            <name name-style="western">
              <surname>Shell</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Fahy</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Xing</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Integrating large language models into medication management in remote healthcare: current applications, challenges, and future prospects</article-title>
          <source>Systems</source>
          <year>2025</year>
          <month>04</month>
          <day>10</day>
          <volume>13</volume>
          <issue>4</issue>
          <fpage>281</fpage>
          <pub-id pub-id-type="doi">10.3390/systems13040281</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Atchinson</surname>
              <given-names>BK</given-names>
            </name>
            <name name-style="western">
              <surname>Fox</surname>
              <given-names>DM</given-names>
            </name>
          </person-group>
          <article-title>The politics of the Health Insurance Portability and Accountability Act</article-title>
          <source>Health Aff (Millwood)</source>
          <year>1997</year>
          <volume>16</volume>
          <issue>3</issue>
          <fpage>146</fpage>
          <lpage>150</lpage>
          <pub-id pub-id-type="doi">10.1377/hlthaff.16.3.146</pub-id>
          <pub-id pub-id-type="medline">9141331</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>AE</given-names>
            </name>
            <name name-style="western">
              <surname>Pollard</surname>
              <given-names>TJ</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lehman</surname>
              <given-names>LH</given-names>
            </name>
            <name name-style="western">
              <surname>Feng</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ghassemi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Moody</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Szolovits</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Celi</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Mark</surname>
              <given-names>RG</given-names>
            </name>
          </person-group>
          <article-title>MIMIC-III, a freely accessible critical care database</article-title>
          <source>Sci Data</source>
          <year>2016</year>
          <volume>3</volume>
          <fpage>160035</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/sdata.2016.35"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/sdata.2016.35</pub-id>
          <pub-id pub-id-type="medline">27219127</pub-id>
          <pub-id pub-id-type="pii">sdata201635</pub-id>
          <pub-id pub-id-type="pmcid">PMC4878278</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>AEW</given-names>
            </name>
            <name name-style="western">
              <surname>Bulgarelli</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Gayles</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shammout</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Horng</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pollard</surname>
              <given-names>TJ</given-names>
            </name>
            <name name-style="western">
              <surname>Hao</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Moody</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Gow</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Lehman</surname>
              <given-names>LWH</given-names>
            </name>
            <name name-style="western">
              <surname>Celi</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Mark</surname>
              <given-names>RG</given-names>
            </name>
          </person-group>
          <article-title>MIMIC-IV, a freely accessible electronic health record dataset</article-title>
          <source>Sci Data</source>
          <year>2023</year>
          <month>01</month>
          <day>03</day>
          <volume>10</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <pub-id pub-id-type="doi">10.1038/s41597-022-01899-x</pub-id>
          <pub-id pub-id-type="medline">36596836</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41597-022-01899-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC9810617</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>TF</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Thirunavukarasu</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DSW</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Large language models in health care: development, applications, and challenges</article-title>
          <source>Health Care Sci</source>
          <year>2023</year>
          <month>08</month>
          <volume>2</volume>
          <issue>4</issue>
          <fpage>255</fpage>
          <lpage>263</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38939520"/>
          </comment>
          <pub-id pub-id-type="doi">10.1002/hcs2.61</pub-id>
          <pub-id pub-id-type="medline">38939520</pub-id>
          <pub-id pub-id-type="pii">HCS261</pub-id>
          <pub-id pub-id-type="pmcid">PMC11080827</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Xia</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Qin</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Poon</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>T-Y</given-names>
            </name>
          </person-group>
          <article-title>BioGPT: generative pre-trained transformer for biomedical text generation and mining</article-title>
          <source>Brief Bioinform</source>
          <year>2022</year>
          <month>11</month>
          <day>19</day>
          <volume>23</volume>
          <issue>6</issue>
          <fpage>bbac409</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/bib/article-lookup/doi/10.1093/bib/bbac409"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bib/bbac409</pub-id>
          <pub-id pub-id-type="medline">36156661</pub-id>
          <pub-id pub-id-type="pii">6713511</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Labrak</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Bazoge</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Morin</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Gourraud</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Rouvier</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Dufour</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>BioMistral: a collection of open-source pretrained large language models for medical domains</article-title>
          <source>arXiv. Preprint posted online on February 15, 2024</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2402.10373"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2402.10373</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yagnik</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Jhaveri</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sharma</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Pila</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Medlm: exploring language models for medical question answering systems</article-title>
          <source>arXiv. Preprint posted online on January 21, 2024</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2401.11389"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2401.11389</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Adhikarla</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Yan</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Davison</surname>
              <given-names>BD</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Fu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>A generalist vision-language foundation model for diverse biomedical tasks</article-title>
          <source>Nat Med</source>
          <year>2024</year>
          <volume>30</volume>
          <issue>11</issue>
          <fpage>3129</fpage>
          <lpage>3141</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-03185-2</pub-id>
          <pub-id pub-id-type="medline">39112796</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-03185-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC12581140</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mistral</surname>
              <given-names>AI</given-names>
            </name>
          </person-group>
          <article-title>mistralai / Mistral-Small-24B-Instruct-2501</article-title>
          <source>Hugging Face</source>
          <year>2025</year>
          <access-date>2026-07-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://huggingface.co/mistralai/Mistral-Small-24B-Instruct-2501">https://huggingface.co/mistralai/Mistral-Small-24B-Instruct-2501</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Abdin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Aneja</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Behl</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Bubeck</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Eldan</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Gunasekar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Harrison</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hewett</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Javaheripi</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Phi4 technical report</article-title>
          <source>arXiv. Preprint posted online on December 12, 2024</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2412.08905</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Almazrouei</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Alobeidli</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Alshamsi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cappelli</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cojocaru</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Debbah</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Goffinet</surname>
              <given-names>É</given-names>
            </name>
            <name name-style="western">
              <surname>Hesslow</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Launay</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Malartic</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Mazzotta</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Noune</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Penedo</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>The Falcon series of open language models</article-title>
          <source>arXiv. Preprint posted online on November 28, 2023</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2311.16867</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grattafiori</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dubey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Jauhri</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pandey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kadian</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Al-Dahle</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Letman</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mathur</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schelten</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Vaughan</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>The Llama 3 herd of models</article-title>
          <source>arXiv. Preprint posted online on July 31, 2024</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2407.21783</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sallinen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Solergibert</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Boyé</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Dupont-Roc</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Theimer-Lienhard</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Boisson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Bernath</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Hadhri</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Llama-3-Meditron: an open-weight suite of medical LLMs based on Llama-3.1</article-title>
          <year>2025</year>
          <conf-name>Workshop on Large Language Models and Generative AI for Health at AAAI 2025</conf-name>
          <conf-date>February 25 to March 4, 2025</conf-date>
          <conf-loc>Philadelphia, PA</conf-loc>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lewis</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Perez</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Piktus</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Petroni</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Karpukhin</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Goyal</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Küttler</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lewis</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Yih</surname>
              <given-names>WT</given-names>
            </name>
          </person-group>
          <article-title>Retrieval-augmented generation for knowledge-intensive NLP tasks</article-title>
          <source>Adv Neural Inf Process Syst</source>
          <year>2020</year>
          <volume>33</volume>
          <fpage>9459</fpage>
          <lpage>9474</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://papers.nips.cc/paper_files/paper/2020/file/6b493230205f780e1bc26945df7481e5-Paper.pdf"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wolf</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Debut</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Sanh</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Chaumond</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Delangue</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Moi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cistac</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Rault</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Louf</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>HuggingFace's Transformers: state-of-the-art natural language processing</article-title>
          <source>arXiv. Preprint posted online on October 9, 2019</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.1910.03771</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Loftsson</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Kulynych</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Kaabachi</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Raisaro</surname>
              <given-names>JL</given-names>
            </name>
          </person-group>
          <article-title>Accelerating clinical text annotation in underrepresented languages: a case study on text de-identification</article-title>
          <source>Stud Health Technol Inform</source>
          <year>2024</year>
          <volume>316</volume>
          <fpage>853</fpage>
          <lpage>857</lpage>
          <pub-id pub-id-type="doi">10.3233/SHTI240546</pub-id>
          <pub-id pub-id-type="medline">39176927</pub-id>
          <pub-id pub-id-type="pii">SHTI240546</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>DesRoches</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Wachenheim</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Ameling</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Cibildak</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cibotti</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Dong</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Drane</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Henderson</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Hurwitz</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Meddings</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Naimark</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>O'Donnell</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Winger</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Winnay</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Young</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wolff</surname>
              <given-names>JL</given-names>
            </name>
          </person-group>
          <article-title>Identifying, engaging, and supporting care partners in primary care settings: a portal-based intervention</article-title>
          <source>BMC Prim Care</source>
          <year>2025</year>
          <volume>26</volume>
          <issue>1</issue>
          <fpage>356</fpage>
          <pub-id pub-id-type="doi">10.1186/s12875-025-03059-7</pub-id>
          <pub-id pub-id-type="medline">41219954</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12875-025-03059-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC12606955</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hauer</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Shi</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Kondrak</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Don't trust ChatGPT when your question is not in English: a study of multilingual abilities and types of LLMs</article-title>
          <year>2023</year>
          <conf-name>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</conf-name>
          <conf-date>December 6-10, 2023</conf-date>
          <conf-loc>Singapore</conf-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>7915</fpage>
          <lpage>7927</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.491</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vadlapati</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Multilingual prompting in LLMs: investigating the accuracy and performance</article-title>
          <source>Int J Sci Res Eng Manag</source>
          <year>2023</year>
          <volume>08</volume>
          <issue>12</issue>
          <fpage>1</fpage>
          <lpage>7</lpage>
          <pub-id pub-id-type="doi">10.55041/IJSREM17694</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>CY</given-names>
            </name>
          </person-group>
          <article-title>ROUGE: a package for automatic evaluation of summaries</article-title>
          <year>2004</year>
          <conf-name>Workshop on Text Summarization Branches Out, Post-Conference Workshop of ACL 2004</conf-name>
          <conf-date>July 25-26, 2004</conf-date>
          <conf-loc>Barcelona, Spain</conf-loc>
          <fpage>74</fpage>
          <lpage>81</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://aclanthology.org/W04-1013/"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Martin</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Muller</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Suarez</surname>
              <given-names>PO</given-names>
            </name>
            <name name-style="western">
              <surname>Dupont</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Romary</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>de La Clergerie</surname>
              <given-names>ÉV</given-names>
            </name>
            <name name-style="western">
              <surname>Sagot</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Seddah</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>CamemBERT: a tasty French language model</article-title>
          <year>2020</year>
          <conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</conf-name>
          <conf-date>July 5-10, 2020</conf-date>
          <conf-loc>Online</conf-loc>
          <fpage>7203</fpage>
          <lpage>7219</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.645</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Textual similarity as a key metric in machine translation quality estimation</article-title>
          <source>arXiv. Preprint posted online on June 11, 2024</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2406.07440</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Long</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Dai</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>mGTE: generalized long-context text representation and reranking models for multilingual text retrieval</article-title>
          <source>arXiv. Preprint posted online on July 29, 2024</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2407.19669</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Xiao</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Lian</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>BGE M3-Embedding: multi-lingual, multi-functionality, multi-granularity text embeddings through self-knowledge distillation</article-title>
          <source>arXiv. Preprint posted online on February 5, 2024</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2402.03216"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2402.03216</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nussbaum</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Morris</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>Duderstadt</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Mulyar</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Nomic Embed: training a reproducible long context text embedder</article-title>
          <source>arXiv. Preprint posted online on February 2, 2024</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2402.01613"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2402.01613</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kandel</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Moles</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Application de l’indice de flesch à la langue francaise</article-title>
          <source>Cahiers Etudes de Radio-Television</source>
          <year>1958</year>
          <volume>19</volume>
          <fpage>253</fpage>
          <lpage>274</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jeanlucmichel.com/Distanciation/Moles.bibliographie.html"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gwet</surname>
              <given-names>KL</given-names>
            </name>
          </person-group>
          <source>Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters</source>
          <year>2014</year>
          <publisher-loc>Gaithersburg, Maryland</publisher-loc>
          <publisher-name>Advanced Analytics, LLC</publisher-name>
          <fpage>1</fpage>
          <lpage>52</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chua</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Yao</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Learning natural language constraints for safe reinforcement learning of language agents</article-title>
          <source>arXiv. Preprint posted online on April 4, 2025</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2504.03185"/>
          </comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2504.03185</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ganzinger</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kunz</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Fuchs</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Lyu</surname>
              <given-names>CK</given-names>
            </name>
            <name name-style="western">
              <surname>Loos</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Dugas</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pausch</surname>
              <given-names>TM</given-names>
            </name>
          </person-group>
          <article-title>Automated generation of discharge summaries: leveraging large language models with clinical data</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <month>05</month>
          <day>12</day>
          <volume>15</volume>
          <issue>1</issue>
          <fpage>16466</fpage>
          <pub-id pub-id-type="doi">10.1038/s41598-025-01618-7</pub-id>
          <pub-id pub-id-type="medline">40355506</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-025-01618-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC12069548</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Williams</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Bains</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Patel</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Lucas</surname>
              <given-names>AN</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Miao</surname>
              <given-names>BY</given-names>
            </name>
            <name name-style="western">
              <surname>Butte</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Kornblith</surname>
              <given-names>AE</given-names>
            </name>
          </person-group>
          <article-title>Evaluating large language models for drafting emergency department discharge summaries</article-title>
          <source>medRxiv. Preprint posted online on April 04, 2024</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38633805"/>
          </comment>
          <pub-id pub-id-type="doi">10.1101/2024.04.03.24305088</pub-id>
          <pub-id pub-id-type="medline">38633805</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref40">
        <label>40</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sellergren</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kazemzadeh</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jaroensri</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Kiraly</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Traverse</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kohlberger</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jamil</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hughes</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lau</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Medgemma technical report</article-title>
          <source>arXiv. Preprint posted online on July 7, 2025</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2507.05201</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
