<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="review-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id>
      <journal-title>Journal of Medical Internet Research</journal-title>
      <issn pub-type="epub">1438-8871</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v28i1e98184</article-id>
      <article-id pub-id-type="pmid">42618044</article-id>
      <article-id pub-id-type="doi">10.2196/98184</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Review</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Review</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>The Reliability of Human Evaluation of Large Language Models in Health Care Settings: Scoping Review</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Brini</surname>
            <given-names>Stefano</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Ogunbowale</surname>
            <given-names>Oluwatobilola</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Yang</surname>
            <given-names>Euijun</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0008-3589-8152</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Ko</surname>
            <given-names>Siyeon</given-names>
          </name>
          <degrees>MPH</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-6610-6751</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Woo</surname>
            <given-names>Hyekyung</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Department of Health Administration</institution>
            <institution>College of Nursing &#38; Health</institution>
            <institution>Kongju National University</institution>
            <addr-line>56 Gongjudaehak-ro</addr-line>
            <addr-line>Gongju, Chungcheongnam-do, 32588</addr-line>
            <country>Republic of Korea</country>
            <phone>82 41 850 0328</phone>
            <email>hkwoo@kongju.ac.kr</email>
          </address>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-5489-3404</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Department of Health Administration</institution>
        <institution>College of Nursing &#38; Health</institution>
        <institution>Kongju National University</institution>
        <addr-line>Gongju, Chungcheongnam-do</addr-line>
        <country>Republic of Korea</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Institute of Health and Environment</institution>
        <institution>Kongju National University</institution>
        <addr-line>Gongju, Chungcheongnam-do</addr-line>
        <country>Republic of Korea</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Hyekyung Woo <email>hkwoo@kongju.ac.kr</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>19</day>
        <month>8</month>
        <year>2026</year>
      </pub-date>
      <volume>28</volume>
      <elocation-id>e98184</elocation-id>
      <history>
        <date date-type="received">
          <day>14</day>
          <month>4</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>12</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>6</day>
          <month>8</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>7</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Euijun Yang, Siyeon Ko, Hyekyung Woo. Originally published in the Journal of Medical Internet Research (https://www.jmir.org), 19.08.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on https://www.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.jmir.org/2026/1/e98184" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Integration of large language models (LLMs) into health care has accelerated rapidly, yet reliability concerns pose potential risks to patient safety. Although human evaluation has been widely used as an important approach for assessing LLM reliability, a systematic understanding of how such evaluations have been operationalized across studies remains limited.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to characterize the current landscape of human evaluation frameworks for LLM reliability in health care and to identify similarities and differences between the clinical and public health domains.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>In line with the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews) guidelines, PubMed, Web of Science, the Cochrane Library, CINAHL, and Google Scholar were searched for studies published from January 2016 to July 2025. Eligible studies were English-language original research conducted in health care settings that assessed the reliability of LLM-generated responses through human evaluation. Key exclusion criteria were studies without human evaluation and studies focused primarily on LLM model selection, performance optimization, or technical development. Extracted data were analyzed across 3 dimensions: what was evaluated, who evaluated, and how evaluation was conducted. Reported methodological limitations were also categorized and compared between the clinical and public health domains.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Of the 4347 records identified, 71 studies were included in the final analysis (clinical, n=26; public health, n=45). Six reliability indicators were used: accuracy, relevance, completeness, clarity, safety, and consistency. The clinical domain more frequently assessed guideline concordance, internal consistency, and structural coherence, whereas the public health domain more frequently assessed understandability, harm potential, and repeat response consistency. Single-specialty clinicians were the most common evaluators in both domains, although mixed evaluator panels were observed only in the public health domain. Evaluator panels generally consisted of 5 or fewer members. Five-point Likert scales and researcher-defined rubrics were commonly used evaluation approaches in both domains. Key methodological limitations included evaluator subjectivity, nonstandardized indicators, and limited evaluation scope and settings.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>To our knowledge, this is the first review to systematically examine how human evaluations of LLM reliability have been conducted across health care. The focus of reliability evaluation differed across domains, with clinical evaluations giving relatively greater attention to clinical validity and logical rigor, whereas public health evaluations gave relatively greater attention to understandability, practical use, and safe use. These differences suggest that the reliability of health care LLMs is difficult to evaluate adequately using a single universal standard. In addition, the methodological limitations identified in this review indicate that current human evaluation approaches are insufficiently standardized. Therefore, future evaluations of health care LLM reliability need to be guided by standardized evaluation frameworks that reflect domain-specific contexts and encompass indicator definitions, judgment criteria, evaluator guidance, and evaluation procedures.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>large language models</kwd>
        <kwd>artificial intelligence</kwd>
        <kwd>human evaluation</kwd>
        <kwd>evaluation framework</kwd>
        <kwd>reliability</kwd>
        <kwd>health care</kwd>
        <kwd>scoping review</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <sec>
        <title>Rationale</title>
        <p>Large language models (LLMs) are AI technologies that generate human-like natural language responses by learning vast amounts of text data, and their applications have been rapidly expanding across various fields [<xref ref-type="bibr" rid="ref1">1</xref>]. In the health care domain, the potential use of LLMs is under active consideration, including for clinical decision-making support and provision of health information [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. LLMs potentially enhance the accessibility and use of health information not only for health care professionals but also for patients and the general public [<xref ref-type="bibr" rid="ref4">4</xref>]. However, concerns regarding the reliability of LLMs have also increased alongside these potential applications. Recent studies have reported that LLMs may generate recommendations inconsistent with clinical guidelines and provide inconsistent information due to hallucinations and variability across responses [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. In addition, the possibility that LLMs may provide inappropriate or unsafe medical advice has been empirically demonstrated [<xref ref-type="bibr" rid="ref9">9</xref>], suggesting that these reliability concerns may extend beyond information quality and contribute to inappropriate health behaviors, distorted clinical judgment, and increased patient safety risks [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. Accordingly, systematic evaluation frameworks are increasingly needed to assess the reliability of LLM-generated responses in health care [<xref ref-type="bibr" rid="ref12">12</xref>].</p>
        <p>In health care settings, LLM applications span both the public health and clinical domains, which serve distinct purposes and populations [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. In the public health domain, LLMs provide health information, support patient education, and facilitate health counseling for patients and the general public, primarily supporting disease prevention and health promotion. In the clinical domain, LLMs support physician decision-making in terms of diagnosis, treatment, and examination, and assist clinical research and clinical documentation in contexts directly related to patient care. The two domains differ in terms of service objectives, risk levels, scope of accountability, and regulatory environments [<xref ref-type="bibr" rid="ref15">15</xref>], suggesting that the reliability evaluation criteria should also be contextually differentiated. However, prior studies have tended to evaluate LLM reliability without sufficiently considering the domain-specific characteristics, thereby limiting the interpretability and applicability of the findings [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>].</p>
        <p>The reliability of LLM-generated responses is primarily verified via both benchmark and human evaluation. Although the former enables objective comparisons based on the correspondence with predefined “correct” answers [<xref ref-type="bibr" rid="ref18">18</xref>], it does not fully capture the contextual appropriateness and practical use required in real-world health care [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. Consequently, human evaluation, in which human evaluators directly assess the trustworthiness of LLM-generated responses based on one or more evaluation indicators, has been increasingly recognized as a core method for evaluation of LLM reliability in the health care field [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. Although prior reviews have examined LLM evaluation in health care, they have primarily focused on application areas and performance outcomes [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. However, studies that have systematically examined how human evaluation was conducted in practice, including the indicators used, evaluators involved, and evaluation procedures, remain limited.</p>
        <p>Therefore, this review aimed to conduct a scoping review of studies in which humans evaluated the reliability of LLM-generated responses in health care. Specifically, we systematically analyzed the evaluation indicators, evaluator characteristics, and evaluation approaches and examined differences between the clinical and public health domains. In addition, we synthesized the methodological limitations reported in the included studies and identified key considerations for the future development of reliability evaluation frameworks. These findings are expected to provide foundational evidence to support the standardization of LLM reliability evaluation and improve comparability across studies in health care contexts.</p>
      </sec>
      <sec>
        <title>Objectives</title>
        <p>This review posed the following research questions:</p>
        <list list-type="order">
          <list-item>
            <p>What evaluation indicators, evaluator characteristics, and evaluation approaches have been used in human evaluation-based LLM reliability evaluations in the health care domain?</p>
          </list-item>
          <list-item>
            <p>How do these evaluation frameworks differ between the clinical and public health domains?</p>
          </list-item>
          <list-item>
            <p>What limitations are commonly observed in terms of human evaluation-based LLM reliability evaluations?</p>
          </list-item>
        </list>
      </sec>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Protocol and Registration</title>
        <p>This review is a scoping review of studies that used human evaluators to evaluate the reliability of LLM-generated responses in the health care domain. The review adhered to the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews) guidelines to enhance transparency and reproducibility in reporting [<xref ref-type="bibr" rid="ref23">23</xref>]. Search reporting was additionally informed by the PRISMA-S (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Searching) recommendations to strengthen the transparency and reproducibility of the search process [<xref ref-type="bibr" rid="ref24">24</xref>]. The PRISMA-ScR reporting checklist is shown in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. A formal review protocol was not publicly registered.</p>
      </sec>
      <sec>
        <title>Information Sources</title>
        <p>A systematic literature search was conducted separately in PubMed, Web of Science, the Cochrane Library, and CINAHL. Google Scholar was used as a supplementary source to screen the first 300 relevance-ranked results for arXiv preprints, consistent with previous recommendations [<xref ref-type="bibr" rid="ref25">25</xref>]. The publication period was restricted from January 1, 2016, to July 31, 2025. The final search was conducted on June 29, 2026, and no subsequent search update was performed. To further improve search comprehensiveness, backward and forward citation searching was conducted for the included studies.</p>
      </sec>
      <sec>
        <title>Search</title>
        <p>The search strategy was developed with reference to the key search terms and search methods used in previous studies on LLMs, generative AI, and health care [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. Search terms were organized based on the Population, Concept, and Context framework recommended for scoping reviews [<xref ref-type="bibr" rid="ref28">28</xref>]. As no specific participant population was required, the Concept was defined as human evaluation of LLM reliability and the context as health care settings. The search terms were grouped into 4 core concepts: LLMs, evaluation, reliability, and health care. Controlled vocabulary and free-text terms were used as appropriate for each source and combined using Boolean operators. Only a publication date restriction was applied during the searches, with no additional restrictions based on study design or publication type. The search strategy was iteratively reviewed and refined through discussions among the research team. The full search strategies for each database are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p>
      </sec>
      <sec>
        <title>Eligibility Criteria</title>
        <p>Eligibility criteria were defined according to the review objective (<xref ref-type="boxed-text" rid="box1">Textbox 1</xref>). Original studies published in English that assessed, through human evaluation, the reliability of responses generated by LLMs or LLM-based chatbots in health care were included, provided that the full text was accessible. Studies that conducted both human and automated evaluation were included when human evaluation results were reported separately, but the analysis in this review was limited to human evaluation data.</p>
        <p>Studies that did not conduct human evaluation of LLM-generated responses, studies primarily focused on LLM model selection or performance optimization, and studies focused on the development of LLMs or related technologies themselves were excluded.</p>
        <boxed-text id="box1" position="float">
          <title>Eligibility criteria.</title>
          <p>Inclusion criteria</p>
          <list list-type="bullet">
            <list-item>
              <p>Studies conducted in the health care domain, encompassing both clinical and public health contexts</p>
            </list-item>
            <list-item>
              <p>Original studies that assessed the reliability of large language model (LLM)–generated responses via human evaluation, including studies using both human and automated evaluation if human evaluation results were reported separately</p>
            </list-item>
            <list-item>
              <p>Studies published in English with accessible full texts</p>
            </list-item>
          </list>
          <p>Exclusion criteria</p>
          <list list-type="bullet">
            <list-item>
              <p>Studies that were not original research (eg, reviews, editorials, commentaries, perspectives, or opinion papers)</p>
            </list-item>
            <list-item>
              <p>Studies that did not involve human evaluation of LLM-generated responses (eg, benchmark-only studies or studies using automated performance evaluation alone)</p>
            </list-item>
            <list-item>
              <p>Studies primarily focused on selecting or optimizing LLMs for specific tasks</p>
            </list-item>
            <list-item>
              <p>Studies primarily focused on the technical development of LLMs or related technologies without evaluating the reliability of LLM-generated responses in health care</p>
            </list-item>
            <list-item>
              <p>Studies published in languages other than English or for which the full text was unavailable</p>
            </list-item>
          </list>
        </boxed-text>
      </sec>
      <sec>
        <title>Selection of Sources of Evidence</title>
        <p>Two researchers (EY and SK) independently conducted the study selection process using Rayyan (Rayyan Systems), including title and abstract screening and full-text review, after duplicate removal [<xref ref-type="bibr" rid="ref29">29</xref>]. After initial screening based on titles and abstracts, full texts were reviewed for eligibility. Any discrepancies were resolved through discussion and consensus between the 2 reviewers.</p>
      </sec>
      <sec>
        <title>Data Charting Process</title>
        <p>Data from the final included studies were systematically extracted and recorded using a Microsoft Excel spreadsheet. One researcher (EY) initially extracted the data, and another researcher (SK) subsequently reviewed the extracted data for accuracy and consistency. The data charting form was developed by the research team based on the review objective and was iteratively reviewed and refined during the review process. When the classification of the domain, application area, or other data items was unclear, the 2 researchers (EY and SK) classified the items by considering the study objectives, the context of LLM use, and other relevant information reported in the full text. Any discrepancies were resolved through discussion and consensus.</p>
      </sec>
      <sec>
        <title>Data Items</title>
        <p>The extracted data included title, study design, publication year, first author, country of the first author, LLM model, LLM task, domain, application area, evaluation indicators, evaluator characteristics, evaluation approaches, and reported limitations. The complete dataset is presented in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. To further analyze human evaluations of LLM reliability, an analytical framework centered on the core components of evaluation was established. The framework comprised 3 analytical dimensions: what was evaluated (evaluation indicators), who conducted the evaluation (evaluator characteristics), and how the evaluation was performed (evaluation approaches). The operational definitions and key examples for each dimension are presented in <xref ref-type="table" rid="table1">Table 1</xref>.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Core dimensions and operational definitions used to analyze human evaluation of large language model reliability in health care settings.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="250"/>
            <col width="450"/>
            <col width="300"/>
            <thead>
              <tr valign="top">
                <td>Category</td>
                <td>Definition</td>
                <td>Key examples</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Evaluation indicator</td>
                <td>Specific dimensions or attributes of LLM<sup>a</sup>-generated responses used to assess reliability</td>
                <td>Accuracy, clarity</td>
              </tr>
              <tr valign="top">
                <td>Evaluator characteristics</td>
                <td>Characteristics of the human evaluators involved in assessing reliability, including their expertise and professional roles</td>
                <td>Clinicians, nurses, researchers</td>
              </tr>
              <tr valign="top">
                <td>Evaluation approach</td>
                <td>Methods or tools used to measure evaluation indicators</td>
                <td>Likert scales, researcher-defined rubrics</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>LLM: large language model.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Critical Appraisal of Individual Sources of Evidence</title>
        <p>Quality of the identified studies was not assessed as this is not a requirement of scoping reviews.</p>
      </sec>
      <sec>
        <title>Synthesis of Results</title>
        <p>Descriptive analyses involving the calculation of frequencies and percentages were performed using Microsoft Excel. Evaluation indicators were integrated based on conceptual similarities in their names and definitions to derive core indicators and subcriteria. The distributions of evaluation indicators, evaluator characteristics, and evaluation approaches were compared between the clinical and public health domains. Methodological limitations were also categorized and compared across the 2 domains.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Selection of Sources of Evidence</title>
        <p>A total of 4347 records were identified across the 5 databases. After removing 1822 duplicate records, 2525 underwent title and abstract screening, of which 2218 were excluded. The remaining 307 reports were sought for retrieval, and 53 were not retrieved. A total of 254 reports were assessed for eligibility via full-text review. A total of 190 reports were excluded because they focused on model selection or optimization (n=105), did not include human evaluation of generated responses (n=40), focused on technical development (n=33), or were not published in English (n=12). In addition, 12 records were identified through citation searching; 2 reports were not retrieved, and 3 of the 10 reports assessed for eligibility were excluded. Ultimately, 71 studies [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref100">100</xref>] were included, comprising 64 studies [<xref ref-type="bibr" rid="ref31">31</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref61">61</xref>-<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref69">69</xref>-<xref ref-type="bibr" rid="ref100">100</xref>] identified through database searching and 7 studies [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref68">68</xref>] through citation searching. The literature selection process is presented in <xref rid="figure1" ref-type="fig">Figure 1</xref> in accordance with the PRISMA-ScR guidelines.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews) flow diagram illustrating the identification, screening, eligibility assessment, and inclusion of studies through database and citation searching.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e98184_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Characteristics of Sources of Evidence</title>
        <p>Of the finally included studies, 26 studies (36.6%) [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref75">75</xref>-<xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref84">84</xref>,<xref ref-type="bibr" rid="ref86">86</xref>,<xref ref-type="bibr" rid="ref96">96</xref>,<xref ref-type="bibr" rid="ref97">97</xref>] were classified under the clinical domain and 45 studies (63.4%) [<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref81">81</xref>-<xref ref-type="bibr" rid="ref83">83</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref87">87</xref>-<xref ref-type="bibr" rid="ref95">95</xref>,<xref ref-type="bibr" rid="ref98">98</xref>-<xref ref-type="bibr" rid="ref100">100</xref>] under the public health domain. In terms of publication year, 6 studies (8.5%) [<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref72">72</xref>] were published in 2023, 32 studies (45.1%) [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref38">38</xref>-<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref65">65</xref>-<xref ref-type="bibr" rid="ref70">70</xref>, <xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref79">79</xref>,<xref ref-type="bibr" rid="ref82">82</xref>,<xref ref-type="bibr" rid="ref84">84</xref>,<xref ref-type="bibr" rid="ref86">86</xref>, <xref ref-type="bibr" rid="ref87">87</xref>,<xref ref-type="bibr" rid="ref90">90</xref>-<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref99">99</xref>] in 2024, and 33 studies (46.5%) [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref53">53</xref>-<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref76">76</xref>-<xref ref-type="bibr" rid="ref78">78</xref>, <xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref83">83</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref88">88</xref>,<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref95">95</xref>-<xref ref-type="bibr" rid="ref98">98</xref>,<xref ref-type="bibr" rid="ref100">100</xref>] through July 2025, indicating a continued increase in health care LLM reliability evaluation research, with 2025 representing a partial year. Although the search covered literature published from 2016 onward, no eligible studies were identified prior to 2023. The LLM application areas were classified into 5 categories based on the literature: medical question answering, clinical decision support, patient education, medical education, and clinical documentation support [<xref ref-type="bibr" rid="ref101">101</xref>]. <xref rid="figure2" ref-type="fig">Figure 2</xref> presents the distribution of included studies by publication year and LLM application area, illustrating both the increase in study volume and the diversification of application areas over time. In 2023, studies were limited to medical question answering and clinical decision support. In 2024, patient education emerged as an additional application area, while medical question answering and clinical decision support continued to account for a substantial proportion of studies. In 2025, studies covered all 5 application areas, with medical education and clinical documentation support newly identified. Regarding study design, comparative studies accounted for the majority of studies (38/71, 53.5%) [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>-<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref47">47</xref>-<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>-<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref59">59</xref>-<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref77">77</xref>,<xref ref-type="bibr" rid="ref79">79</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref82">82</xref>,<xref ref-type="bibr" rid="ref84">84</xref>-<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref95">95</xref>,<xref ref-type="bibr" rid="ref96">96</xref>,<xref ref-type="bibr" rid="ref99">99</xref>], followed by observational studies (23/71, 32.4%) [<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref65">65</xref>-<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref69">69</xref>-<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref78">78</xref>,<xref ref-type="bibr" rid="ref92">92</xref>-<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref100">100</xref>] and diagnostic accuracy studies (10/71, 14.1%) [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref83">83</xref>,<xref ref-type="bibr" rid="ref97">97</xref>,<xref ref-type="bibr" rid="ref98">98</xref>]. Studies were conducted across North America, Europe, Asia, and Oceania. The United States contributed the largest number of studies (16/71, 22.5%) [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref51">51</xref>-<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref66">66</xref>,<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref79">79</xref>,<xref ref-type="bibr" rid="ref83">83</xref>,<xref ref-type="bibr" rid="ref84">84</xref>,<xref ref-type="bibr" rid="ref92">92</xref>,<xref ref-type="bibr" rid="ref95">95</xref>], followed by Türkiye (12/71, 16.9%) [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref96">96</xref>,<xref ref-type="bibr" rid="ref98">98</xref>], China (9/71, 12.7%) [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref69">69</xref>,<xref ref-type="bibr" rid="ref97">97</xref>,<xref ref-type="bibr" rid="ref99">99</xref>], and Germany (7/71, 9.9%) [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref87">87</xref>,<xref ref-type="bibr" rid="ref88">88</xref>]. In terms of the LLM models evaluated, OpenAI’s ChatGPT was assessed in nearly all studies (70/71, 98.6%) [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref57">57</xref>-<xref ref-type="bibr" rid="ref100">100</xref>]. Google’s Gemini/Bard/SLE models were evaluated in 25 studies (25/71, 35.2%) [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref48">48</xref>-<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref55">55</xref>-<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref77">77</xref>,<xref ref-type="bibr" rid="ref81">81</xref>, <xref ref-type="bibr" rid="ref82">82</xref>,<xref ref-type="bibr" rid="ref84">84</xref>,<xref ref-type="bibr" rid="ref86">86</xref>-<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref95">95</xref>,<xref ref-type="bibr" rid="ref96">96</xref>], Anthropic’s Claude in 9 studies (9/71, 12.7%) [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref77">77</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref95">95</xref>], Microsoft’s Copilot/Bing models in 9 studies (9/71, 12.7%) [<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref86">86</xref>,<xref ref-type="bibr" rid="ref91">91</xref>], and Meta’s Llama-based models in 2 studies (2/71, 2.8%) [<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref89">89</xref>]. Detailed characteristics of the included studies are presented in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Distribution of the included studies by large language model application area and publication year. Blue circles represent studies in the clinical domain, and green circles represent studies in the public health domain. The number within each circle and the circle size indicate the number of studies.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e98184_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Critical Appraisal Within Sources of Evidence</title>
        <p>The risk of bias in studies was not assessed, in line with scoping review methodology.</p>
      </sec>
      <sec>
        <title>Results of Individual Sources of Evidence</title>
        <p>The detailed results reported by each included study are presented in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>, which summarizes the key data extracted from the included studies.</p>
      </sec>
      <sec>
        <title>Synthesis of Results</title>
        <sec>
          <title>Evaluation Indicators</title>
          <p>Across the 71 included studies [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref100">100</xref>], various evaluation indicators were used to assess reliability; however, their terminology and definitions varied considerably. In this review, reliability was not treated as a fixed a priori construct but was operationally defined as the set of evaluation attributes used by human evaluators to determine whether LLMs could be trusted and appropriately used in health care contexts. To address this terminological heterogeneity and inductively derive the components of reliability, all evaluation indicators used in the included studies were extracted verbatim and organized according to their original names. The frequency of each indicator was then calculated, and the indicators were reviewed in descending order of reporting frequency. Lower-frequency indicators were integrated into higher-frequency indicators when they were conceptually similar or when their definitions fell within the conceptual scope of the higher-frequency indicators. In contrast, indicators that remained specific to particular studies after examination of their definitions, as well as indicators that were too broad to be classified as a single evaluation concept, were categorized as “Other.” The analysis focused on indicators that could be clearly classified. Through this process, 6 core indicators were identified: accuracy, relevance, completeness, clarity, safety, and consistency. The original indicator names and their classification by core indicator are presented in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. The definitions and evaluation criteria of the indicators integrated into each core indicator were compared and synthesized to derive common subcriteria and corresponding definitions, as presented in <xref ref-type="table" rid="table2">Table 2</xref>.</p>
          <p><xref ref-type="table" rid="table3">Table 3</xref> presents the frequencies of the 6 core reliability evaluation indicators and their subcriteria across the clinical and public health domains. As a single evaluation indicator may encompass multiple subcriteria, the sum of the percentages within each domain exceeds 100%. For example, if a study defined accuracy as a concept that incorporated both guideline concordance and currency, that study was counted under both subcriteria [<xref ref-type="bibr" rid="ref75">75</xref>]. In addition, 2 studies [<xref ref-type="bibr" rid="ref96">96</xref>,<xref ref-type="bibr" rid="ref97">97</xref>] in the clinical domain and 3 studies [<xref ref-type="bibr" rid="ref98">98</xref>-<xref ref-type="bibr" rid="ref100">100</xref>] in the public health domain that did not report definitions of the evaluation indicators were classified separately as “Not Reported.”</p>
          <p>Overall, accuracy was the most frequently used core evaluation indicator in both domains, although differences were observed in the relative emphasis placed on specific subcriteria. In the clinical domain, guideline concordance was used as a major evaluation component alongside factual correctness, whereas in the public health domain, evaluation primarily focused on factual correctness. Evidence and source validity were assessed at similar frequencies across the 2 domains, while studies explicitly incorporating currency were limited in both domains.</p>
          <p>Relevance was assessed primarily via query-response alignment and actionability in the clinical domain, with contextual appropriateness additionally used in a subset of studies. In the public health domain, actionability was the most frequently used subcriterion, followed by query-response alignment.</p>
          <p>In terms of completeness, studies in the clinical domain mainly focused on core element coverage, assessing whether responses addressed the clinically required components. In the public health domain, core element coverage was the most frequently assessed subcriterion, followed by information sufficiency, whereas detail inclusion was infrequently assessed.</p>
          <p>Clarity was most frequently assessed through understandability, particularly in the public health domain. In contrast, structural coherence was more frequently assessed in the clinical domain than in the public health domain.</p>
          <p>In terms of safety, studies evaluating both harm potential and confusion potential were observed in both domains. Both were reported approximately 3 times as frequently in the public health domain as in the clinical domain (harm potential: clinical, 2/26, 7.69%; public health, 10/45, 22.22%; confusion potential: clinical, 1/26, 3.85%; public health, 5/45, 11.11%). Warning provision was assessed in only 1 clinical study and was not assessed in the public health domain.</p>
          <p>Patterns also differed for Consistency. Repeat response consistency was more frequently assessed in the public health domain, whereas internal consistency was assessed more than 3 times as frequently in the clinical domain (4/26, 15.38%) as in the public health domain (2/45, 4.44%).</p>
          <table-wrap position="float" id="table2">
            <label>Table 2</label>
            <caption>
              <p>Core reliability evaluation indicators and subcriteria used for human evaluation of large language model-generated responses in health care settings, derived from 71 studies<sup>a,b,c</sup>.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="30"/>
              <col width="270"/>
              <col width="400"/>
              <col width="0"/>
              <col width="300"/>
              <thead>
                <tr valign="top">
                  <td colspan="2">Evaluation indicator</td>
                  <td>Definition</td>
                  <td colspan="2">Studies</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td colspan="4">Accuracy</td>
                  <td>[<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref85">85</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Factual correctness</td>
                  <td>Factually accurate and free from errors or distortions</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Guideline concordance</td>
                  <td>Consistent with established clinical guidelines, the medical literature, or health policies</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Currency</td>
                  <td>Reflecting up-to-date medical knowledge, public health information, or policy changes</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Evidence and source validity</td>
                  <td>Supported by credible and appropriate evidence or references</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td colspan="4">Relevance</td>
                  <td>[<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>-<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<break/><xref ref-type="bibr" rid="ref56">56</xref>-<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref74">74</xref>-<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref78">78</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref84">84</xref>-<xref ref-type="bibr" rid="ref90">90</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Query-response alignment</td>
                  <td>Directly addressing the intent of the query</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Contextual appropriateness</td>
                  <td>Appropriately tailored to the given context, including user characteristics or situational factors</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Actionability</td>
                  <td>Provision of useful and actionable guidance applicable to real-world decision-making or behavior</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td colspan="4">Completeness</td>
                  <td>[<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref55">55</xref>-<xref ref-type="bibr" rid="ref61">61</xref>,<break/><xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref68">68</xref>-<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref74">74</xref>-<xref ref-type="bibr" rid="ref77">77</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref85">85</xref>-<xref ref-type="bibr" rid="ref88">88</xref>,<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Core element coverage</td>
                  <td>Covering all essential components required to address the query</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Information sufficiency</td>
                  <td>Provision of sufficient and nonomissive information</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Detail inclusion</td>
                  <td>With detailed and in-depth information beyond superficial descriptions</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td colspan="4">Clarity</td>
                  <td>[<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref49">49</xref>-<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<break/><xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref73">73</xref>-<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref82">82</xref>,<xref ref-type="bibr" rid="ref86">86</xref>-<xref ref-type="bibr" rid="ref90">90</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Understandability</td>
                  <td>Clear and easily understandable by the intended audience</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Structural coherence</td>
                  <td>Logically organized and well-structured</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td colspan="4">Safety</td>
                  <td>[<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref78">78</xref>,<xref ref-type="bibr" rid="ref79">79</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref90">90</xref>,<xref ref-type="bibr" rid="ref92">92</xref>-<xref ref-type="bibr" rid="ref94">94</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Harm potential</td>
                  <td>With information that may pose risks to health or safety</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Confusion potential</td>
                  <td>Absence of misunderstandings, confusion, or unnecessary concern because of ambiguity or inaccuracy</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Warning provision</td>
                  <td>Provision of appropriate warnings, cautions, or risk mitigation guidance</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td colspan="4">Consistency</td>
                  <td>[<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref69">69</xref>,<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref86">86</xref>,<xref ref-type="bibr" rid="ref90">90</xref>,<xref ref-type="bibr" rid="ref95">95</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Repeat response consistency</td>
                  <td>Remaining stable across repeated outputs for the same query</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Internal consistency</td>
                  <td>Logically consistent, without internal contradictions</td>
                  <td colspan="2">
                    <break/>
                  </td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table2fn1">
                <p><sup>a</sup>Indicators and subcriteria were derived through comparative analysis of evaluation frameworks reported across the included studies.</p>
              </fn>
              <fn id="table2fn2">
                <p><sup>b</sup>A single study could use more than one evaluation indicator or subcriterion; therefore, studies may be counted in multiple categories.</p>
              </fn>
              <fn id="table2fn3">
                <p><sup>c</sup>Reference numbers correspond to studies that used each indicator.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
          <table-wrap position="float" id="table3">
            <label>Table 3</label>
            <caption>
              <p>Core reliability evaluation indicators and subcriteria used in human evaluations of large language models across the clinical and public health domains<sup>a</sup>.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="30"/>
              <col width="230"/>
              <col width="130"/>
              <col width="190"/>
              <col width="0"/>
              <col width="140"/>
              <col width="280"/>
              <thead>
                <tr valign="top">
                  <td colspan="2">Evaluation indicator</td>
                  <td colspan="3">Clinical (n=26)</td>
                  <td colspan="2">Public health (n=45)</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">
                    <break/>
                  </td>
                  <td>Studies, n (%)</td>
                  <td>References</td>
                  <td colspan="2">Studies, n (%)</td>
                  <td>References</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td colspan="7">Accuracy</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Factual correctness</td>
                  <td>16 (61.54)</td>
                  <td>[<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]</td>
                  <td colspan="2">29 (64.44)</td>
                  <td>[<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref74">74</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Guideline concordance</td>
                  <td>7 (26.92)</td>
                  <td>[<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref75">75</xref>-<xref ref-type="bibr" rid="ref80">80</xref>]</td>
                  <td colspan="2">5 (11.11)</td>
                  <td>[<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref81">81</xref>-<xref ref-type="bibr" rid="ref83">83</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Currency</td>
                  <td>2 (7.69)</td>
                  <td>[<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref75">75</xref>]</td>
                  <td colspan="2">2 (4.44)</td>
                  <td>[<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref70">70</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Evidence and source validity</td>
                  <td>4 (15.38)</td>
                  <td>[<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref84">84</xref>]</td>
                  <td colspan="2">6 (13.33)</td>
                  <td>[<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref85">85</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="7">Relevance</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Query-response alignment</td>
                  <td>9 (34.62)</td>
                  <td>[<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref86">86</xref>]</td>
                  <td colspan="2">8 (17.78)</td>
                  <td>[<xref ref-type="bibr" rid="ref47">47</xref>-<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref71">71</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Contextual appropriateness</td>
                  <td>4 (15.38)</td>
                  <td>[<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref78">78</xref>,<xref ref-type="bibr" rid="ref84">84</xref>]</td>
                  <td colspan="2">4 (8.89)</td>
                  <td>[<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref87">87</xref>,<xref ref-type="bibr" rid="ref88">88</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Actionability</td>
                  <td>7 (26.92)</td>
                  <td>[<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref78">78</xref>,<xref ref-type="bibr" rid="ref84">84</xref>]</td>
                  <td colspan="2">13 (28.89)</td>
                  <td>[<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref90">90</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="7">Completeness</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Core element coverage</td>
                  <td>7 (26.92)</td>
                  <td>[<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref75">75</xref>-<xref ref-type="bibr" rid="ref77">77</xref>]</td>
                  <td colspan="2">16 (35.56)</td>
                  <td>[<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref55">55</xref>-<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref69">69</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref87">87</xref>,<xref ref-type="bibr" rid="ref88">88</xref>,<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Information sufficiency</td>
                  <td>3 (11.54)</td>
                  <td>[<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]</td>
                  <td colspan="2">10 (22.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Detail inclusion</td>
                  <td>2 (7.69)</td>
                  <td>[<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref86">86</xref>]</td>
                  <td colspan="2">4 (8.89)</td>
                  <td>[<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="7">Clarity</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Understandability</td>
                  <td>7 (26.92)</td>
                  <td>[<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref86">86</xref>]</td>
                  <td colspan="2">20 (44.44)</td>
                  <td>[<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref49">49</xref>-<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref82">82</xref>,<xref ref-type="bibr" rid="ref87">87</xref>-<xref ref-type="bibr" rid="ref90">90</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Structural coherence</td>
                  <td>4 (15.38)</td>
                  <td>[<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref86">86</xref>]</td>
                  <td colspan="2">1 (2.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref50">50</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="7">Safety</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Harm potential</td>
                  <td>2 (7.69)</td>
                  <td>[<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref79">79</xref>]</td>
                  <td colspan="2">10 (22.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref92">92</xref>-<xref ref-type="bibr" rid="ref94">94</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Confusion potential</td>
                  <td>1 (3.85)</td>
                  <td>[<xref ref-type="bibr" rid="ref37">37</xref>]</td>
                  <td colspan="2">5 (11.11)</td>
                  <td>[<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref90">90</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Warning provision</td>
                  <td>1 (3.85)</td>
                  <td>[<xref ref-type="bibr" rid="ref78">78</xref>]</td>
                  <td colspan="2">0 (0)</td>
                  <td>—<sup>b</sup></td>
                </tr>
                <tr valign="top">
                  <td colspan="7">Consistency</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Repeat response consistency</td>
                  <td>2 (7.69)</td>
                  <td>[<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref75">75</xref>]</td>
                  <td colspan="2">6 (13.33)</td>
                  <td>[<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref69">69</xref>,<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref90">90</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>Internal consistency</td>
                  <td>4 (15.38)</td>
                  <td>[<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref86">86</xref>]</td>
                  <td colspan="2">2 (4.44)</td>
                  <td>[<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref95">95</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">Not reported<sup>c</sup></td>
                  <td>2 (7.69)</td>
                  <td>[<xref ref-type="bibr" rid="ref96">96</xref>,<xref ref-type="bibr" rid="ref97">97</xref>]</td>
                  <td colspan="2">3 (6.67)</td>
                  <td>[<xref ref-type="bibr" rid="ref98">98</xref>-<xref ref-type="bibr" rid="ref100">100</xref>]</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table3fn1">
                <p><sup>a</sup>Percentages within each domain exceed 100% because a single study could be classified under multiple subcriteria within the same evaluation indicator.</p>
              </fn>
              <fn id="table3fn2">
                <p><sup>b</sup>Not available.</p>
              </fn>
              <fn id="table3fn3">
                <p><sup>c</sup>Not reported refers to studies that did not provide definitions for the evaluation indicators used.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
        </sec>
        <sec>
          <title>Evaluator Characteristics</title>
          <p><xref ref-type="table" rid="table4">Table 4</xref> presents the evaluator compositions and panel sizes used in reliability evaluations across the clinical and public health domains.</p>
          <p>In the clinical domain, evaluations involving only single-specialty clinicians accounted for the majority of studies. Panel sizes ranged from 1 to 6 or more evaluators, with panels of 1-2 and 3-5 evaluators being equally the most common. Evaluations involving multispecialty clinicians were identified in only a few studies, and evaluations involving only nonclinicians were limited to 2 studies. Evaluator composition was not clearly reported in 2 studies [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref84">84</xref>].</p>
          <p>In the public health domain, evaluator composition was more diverse. Although evaluations involving only single-specialty clinicians remained the most common, mixed evaluator panels comprising both clinicians and nonclinicians were identified only in the public health domain and across multiple studies. The specific compositions varied across studies and included combinations of clinicians, nurses, midwives, postgraduate students, patients, and laypersons [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref66">66</xref>,<xref ref-type="bibr" rid="ref90">90</xref>]. In addition, an evaluator panel that did not include any clinicians was identified in 1 study.</p>
          <table-wrap position="float" id="table4">
            <label>Table 4</label>
            <caption>
              <p>Evaluator characteristics in human evaluations of large language models across the clinical and public health domains.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="30"/>
              <col width="220"/>
              <col width="130"/>
              <col width="180"/>
              <col width="140"/>
              <col width="300"/>
              <thead>
                <tr valign="top">
                  <td colspan="2">Evaluator characteristics</td>
                  <td colspan="2">Clinical (n=26)</td>
                  <td colspan="2">Public health (n=45)</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">
                    <break/>
                  </td>
                  <td>Studies, n (%)</td>
                  <td>References</td>
                  <td>Studies, n (%)</td>
                  <td>References</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td colspan="6">Clinicians only (single-specialty)</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>1-2</td>
                  <td>7 (26.92)</td>
                  <td>[<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref79">79</xref>]</td>
                  <td>13 (28.89)</td>
                  <td>[<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref69">69</xref>,<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref83">83</xref>,<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref98">98</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3-5</td>
                  <td>7 (26.92)</td>
                  <td>[<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref97">97</xref>]</td>
                  <td>10 (22.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref81">81</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref89">89</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>≥ 6</td>
                  <td>5 (19.23)</td>
                  <td>[<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref77">77</xref>,<xref ref-type="bibr" rid="ref86">86</xref>]</td>
                  <td>9 (20)</td>
                  <td>[<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref82">82</xref>,<xref ref-type="bibr" rid="ref87">87</xref>,<xref ref-type="bibr" rid="ref88">88</xref>,<xref ref-type="bibr" rid="ref93">93</xref>,<xref ref-type="bibr" rid="ref94">94</xref>,<xref ref-type="bibr" rid="ref99">99</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="6">Clinicians only (multispecialties)</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>1-2</td>
                  <td>0 (0)</td>
                  <td>—<sup>a</sup></td>
                  <td>0 (0)</td>
                  <td>—</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3-5</td>
                  <td>1 (3.85)</td>
                  <td>[<xref ref-type="bibr" rid="ref41">41</xref>]</td>
                  <td>2 (4.44)</td>
                  <td>[<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref70">70</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>≥ 6</td>
                  <td>2 (7.69)</td>
                  <td>[<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]</td>
                  <td>0 (0)</td>
                  <td>—</td>
                </tr>
                <tr valign="top">
                  <td colspan="6">Mixed evaluators<sup>b</sup></td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>1-2</td>
                  <td>0 (0)</td>
                  <td>—</td>
                  <td>1 (2.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref95">95</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3-5</td>
                  <td>0 (0)</td>
                  <td>—</td>
                  <td>3 (6.67)</td>
                  <td>[<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref74">74</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>≥6</td>
                  <td>0 (0)</td>
                  <td>—</td>
                  <td>5 (11.11)</td>
                  <td>[<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref66">66</xref>,<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref90">90</xref>,<xref ref-type="bibr" rid="ref100">100</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="6">Others<sup>c</sup></td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>1-2</td>
                  <td>1 (3.85)</td>
                  <td>[<xref ref-type="bibr" rid="ref78">78</xref>]</td>
                  <td>0 (0)</td>
                  <td>—</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3-5</td>
                  <td>1 (3.85)</td>
                  <td>[<xref ref-type="bibr" rid="ref96">96</xref>]</td>
                  <td>1 (2.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref48">48</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>≥6</td>
                  <td>0 (0)</td>
                  <td>—</td>
                  <td>0 (0)</td>
                  <td>—</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">Not specified<sup>d</sup></td>
                  <td>2 (7.69)</td>
                  <td>[<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref84">84</xref>]</td>
                  <td>1 (2.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref58">58</xref>]</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table4fn1">
                <p><sup>a</sup>Not available.</p>
              </fn>
              <fn id="table4fn2">
                <p><sup>b</sup>Mixed evaluators refer to panels comprising both clinicians and nonclinicians within a single study such as patients, caregivers, researchers, or students.</p>
              </fn>
              <fn id="table4fn3">
                <p><sup>c</sup>Others refer to panels comprising only nonclinicians such as anatomists or researchers.</p>
              </fn>
              <fn id="table4fn4">
                <p><sup>d</sup>Not specified refers to studies in which evaluators were identified as clinicians but their clinical specialties were not reported, or in which the evaluator composition was not reported.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
        </sec>
        <sec>
          <title>Evaluation Approach</title>
          <p><xref ref-type="table" rid="table5">Table 5</xref> presents the evaluation approaches used in reliability evaluations across the clinical and public health domains. Because a single study could use multiple evaluation approaches, the sum of percentages within each domain exceeds 100%. For example, if a study used both a 3-point and a 6-point Likert scale, it was counted in both Likert scale categories [<xref ref-type="bibr" rid="ref100">100</xref>].</p>
          <p>In the clinical domain, Likert scale-based evaluation was the most frequently used approach, and the 5-point scale was the most common. Other scale formats included 3-point, 4-point, and 6-point scales. Researcher-defined rubrics were also identified in several studies, whereas validated instruments and checklists were infrequently used. Combined approaches were also identified in a subset of studies.</p>
          <p>In the public health domain, Likert scale-based evaluations and researcher-defined rubrics were both commonly used, with the 5-point scale being the most frequently used Likert scale format. In addition to 3-point and 6-point scales, 10-point and 12-point scales were used, indicating greater variation in scale formats than in the clinical domain. Validated instruments and checklists were infrequently used. Combined approaches were more frequently reported than in the clinical domain. Examples included researcher-defined rubrics combined with 5-point Likert scales, 5-point Likert scales combined with the Global Quality Scale (GQS), and researcher-defined rubrics combined with checklists [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref87">87</xref>].</p>
          <table-wrap position="float" id="table5">
            <label>Table 5</label>
            <caption>
              <p>Evaluation approaches used in human evaluations of large language models across the clinical and public health domains<sup>a</sup>.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="30"/>
              <col width="160"/>
              <col width="130"/>
              <col width="230"/>
              <col width="0"/>
              <col width="150"/>
              <col width="300"/>
              <thead>
                <tr valign="top">
                  <td colspan="2">Evaluation approach</td>
                  <td colspan="3">Clinical (n=26)</td>
                  <td colspan="2">Public health (n=45)</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">
                    <break/>
                  </td>
                  <td>Studies, n (%)</td>
                  <td>References</td>
                  <td colspan="2">Studies, n (%)</td>
                  <td>References</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td colspan="7">Likert scale</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>3-point</td>
                  <td>3 (11.54)</td>
                  <td>[<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref84">84</xref>]</td>
                  <td colspan="2">2 (4.44)</td>
                  <td>[<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref100">100</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>4-point</td>
                  <td>1 (3.85)</td>
                  <td>[<xref ref-type="bibr" rid="ref77">77</xref>]</td>
                  <td colspan="2">0 (0)</td>
                  <td>—<sup>b</sup></td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>5-point</td>
                  <td>11 (42.31)</td>
                  <td>[<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref75">75</xref>,<xref ref-type="bibr" rid="ref80">80</xref>,<xref ref-type="bibr" rid="ref84">84</xref>]</td>
                  <td colspan="2">13 (28.89)</td>
                  <td>[<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref70">70</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref90">90</xref>,<xref ref-type="bibr" rid="ref93">93</xref>,<xref ref-type="bibr" rid="ref99">99</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>6-point</td>
                  <td>2 (7.69)</td>
                  <td>[<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]</td>
                  <td colspan="2">4 (8.89)</td>
                  <td>[<xref ref-type="bibr" rid="ref73">73</xref>,<xref ref-type="bibr" rid="ref90">90</xref>,<xref ref-type="bibr" rid="ref98">98</xref>,<xref ref-type="bibr" rid="ref100">100</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>10-point</td>
                  <td>0 (0)</td>
                  <td>—</td>
                  <td colspan="2">1 (2.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref94">94</xref>]</td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>12-point</td>
                  <td>0 (0)</td>
                  <td>—</td>
                  <td colspan="2">1 (2.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref98">98</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">Researcher-defined rubric<sup>c</sup></td>
                  <td>8 (30.77)</td>
                  <td>[<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref76">76</xref>,<xref ref-type="bibr" rid="ref78">78</xref>,<xref ref-type="bibr" rid="ref86">86</xref>]</td>
                  <td colspan="2">15 (33.33)</td>
                  <td>[<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref61">61</xref>-<xref ref-type="bibr" rid="ref63">63</xref>,<xref ref-type="bibr" rid="ref66">66</xref>,<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref69">69</xref>,<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref82">82</xref>,<xref ref-type="bibr" rid="ref83">83</xref>,<xref ref-type="bibr" rid="ref85">85</xref>,<xref ref-type="bibr" rid="ref95">95</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">Validated instruments<sup>d</sup></td>
                  <td>1 (3.85)</td>
                  <td>[<xref ref-type="bibr" rid="ref96">96</xref>]</td>
                  <td colspan="2">1 (2.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref89">89</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">Combined approaches<sup>e</sup></td>
                  <td>3 (11.54)</td>
                  <td>[<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref97">97</xref>]</td>
                  <td colspan="2">11 (24.44)</td>
                  <td>[<xref ref-type="bibr" rid="ref49">49</xref>-<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref64">64</xref>,<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref87">87</xref>,<xref ref-type="bibr" rid="ref88">88</xref>,<xref ref-type="bibr" rid="ref91">91</xref>,<xref ref-type="bibr" rid="ref92">92</xref>]</td>
                </tr>
                <tr valign="top">
                  <td colspan="2">Checklists</td>
                  <td>1 (3.85)</td>
                  <td>[<xref ref-type="bibr" rid="ref79">79</xref>]</td>
                  <td colspan="2">1 (2.22)</td>
                  <td>[<xref ref-type="bibr" rid="ref81">81</xref>]</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table5fn1">
                <p><sup>a</sup>Percentages within each domain exceed 100% because a single study could be classified under multiple categories when multiple evaluation approaches were used.</p>
              </fn>
              <fn id="table5fn2">
                <p><sup>b</sup>Not available.</p>
              </fn>
              <fn id="table5fn3">
                <p><sup>c</sup>Researcher-defined rubrics refer to evaluation criteria or scoring frameworks constructed by the researchers for the specific purposes of an individual study.</p>
              </fn>
              <fn id="table5fn4">
                <p><sup>d</sup>Validated instruments refer to standardized assessment tools such as DISCERN and the Patient Education Materials Assessment Tool (PEMAT).</p>
              </fn>
              <fn id="table5fn5">
                <p><sup>e</sup>Combined approaches refer to the use of 2 or more evaluation methods within a single study, including Likert scales, researcher-defined rubrics, validated instruments, and checklists.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
        </sec>
      </sec>
      <sec>
        <title>Methodological Challenges in Reliability Evaluation Across Domains</title>
        <p><xref rid="figure3" ref-type="fig">Figure 3</xref> presents the distribution of methodological limitations reported in the clinical (n=26) and public health (n=45) domains, enabling visual comparison of their relative frequencies. Overall, subjectivity in human evaluation was the most frequently reported limitation in both domains. In the clinical domain, insufficient evaluator sample size was the next most frequently reported limitation, whereas restricted evaluator composition and lack of standardized evaluation metrics were relatively common in the public health domain. Insufficient evaluator sample size and limitations of rating scale–based evaluation were reported more frequently in the clinical domain, while restricted evaluator composition and lack of standardized evaluation metrics were more frequently reported in the public health domain. Limited evaluation scope and representativeness were reported at similar frequencies across the 2 domains, while constrained evaluation settings were more frequently reported in the clinical domain. Sensitivity to input formulation was reported in only 1 study in the public health domain. Specific examples of each limitation and a list of relevant studies are provided in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Frequencies of reported methodological limitations in human evaluations of large language model reliability across the clinical and public health domains. Percentages represent the proportion of studies within each domain, and darker colors indicate higher reporting frequencies.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e98184_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>We systematically examined the evaluation indicators, evaluator characteristics, and evaluation approaches used in human evaluations of LLM reliability in health care and compared the clinical and public health domains. Reliability evaluation was centered on 6 core evaluation indicators: accuracy, relevance, completeness, clarity, safety, and consistency. Among the subcriteria, guideline concordance, internal consistency, and structural coherence were more frequently assessed in the clinical domain, whereas understandability, harm potential, and repeat response consistency were more frequently assessed in the public health domain. Clinicians constituted the primary evaluator group in both domains, and evaluator panels generally consisted of 5 or fewer members. Likert scales and researcher-defined rubrics were the most commonly used evaluation approaches. These findings provide practical guidance for identifying the key elements that should be considered when evaluating LLM reliability across health care contexts and for improving the methodological quality of human evaluation.</p>
      </sec>
      <sec>
        <title>Domain-Specific Differences in Evaluation Indicators and Conceptualization of Reliability</title>
        <p>A key finding of this review was that evaluation indicators for LLM reliability differed between the clinical and public health domains, reflecting differences in how reliability is conceptualized across health care contexts.</p>
        <p>In the clinical domain, the validity and contextual appropriateness of LLM responses for real-world clinical judgment were treated as core components of reliability. Because clinical decision-making is influenced by a patient’s symptoms, medical history, comorbidities, and treatment setting, the appropriateness of the same medical information may differ across specific clinical situations [<xref ref-type="bibr" rid="ref102">102</xref>,<xref ref-type="bibr" rid="ref103">103</xref>]. Accordingly, reliability in the clinical domain can be understood as extending beyond factual accuracy to include whether responses appropriately reflect the intent of the question and the patient’s circumstances. Moreover, given that LLM-generated responses may inform real-world decision-making related to diagnosis, treatment, and medication prescribing, and even a single error may result in serious harm [<xref ref-type="bibr" rid="ref104">104</xref>], the evaluation of clinical validity and logical rigor through expert judgment needs to be considered an important aspect. In contrast, in the public health domain, the understandability, practical applicability, and safe use of LLM-generated health information by laypeople were treated as core components of reliability. Laypeople need to understand, evaluate, and use health information when making health-related decisions [<xref ref-type="bibr" rid="ref105">105</xref>], and prior literature has similarly emphasized that user-tailored communication and clear information delivery are important values of LLM applications in the public health domain [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref106">106</xref>]. Laypeople, particularly those with limited health literacy, may have difficulty evaluating the credibility of conflicting or potentially harmful health information, while inconsistent or confusing responses may lead to inappropriate decisions and actions [<xref ref-type="bibr" rid="ref107">107</xref>,<xref ref-type="bibr" rid="ref108">108</xref>]. Therefore, LLM reliability evaluation in the public health domain needs to consider user-centered aspects reflecting understandability, practical use, and safe use.</p>
        <p>These findings highlight the limitations of applying a single uniform set of evaluation criteria to assess LLM reliability across different health care contexts. Reliability may not represent a universal fixed attribute and may instead be interpreted as a context-dependent concept, with evaluation criteria varying according to purpose, level of risk, and user characteristics. This interpretation is also consistent with previous research suggesting that the trustworthiness of medical AI should be understood in relation to its purpose of use and context of use [<xref ref-type="bibr" rid="ref109">109</xref>,<xref ref-type="bibr" rid="ref110">110</xref>]. Therefore, future evaluations of LLM reliability in health care may require greater consideration of domain-specific evaluation frameworks that reflect such contextual differences.</p>
      </sec>
      <sec>
        <title>Conceptual Inconsistency and the Need for Standardization</title>
        <p>Beyond these domain-specific differences, conceptual inconsistency in evaluation indicators was identified as a common challenge across both domains. Even when the same indicator terminology was used, its meaning and scope of application varied considerably across studies. For example, indicators bearing the same label of “usefulness” were classified differently in this review as relevance, completeness, or clarity, depending on the definitions and evaluation criteria provided in each study [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref74">74</xref>,<xref ref-type="bibr" rid="ref90">90</xref>]. In addition, some studies did not provide explicit operational definitions for the indicators used [<xref ref-type="bibr" rid="ref96">96</xref>-<xref ref-type="bibr" rid="ref100">100</xref>]. Such conceptual inconsistency in evaluation indicators may hinder direct comparisons across studies and limit the consistent accumulation of evidence on the reliability of health care LLMs [<xref ref-type="bibr" rid="ref111">111</xref>,<xref ref-type="bibr" rid="ref112">112</xref>]. Previous studies have also consistently noted the lack of consensus regarding evaluation dimensions and criteria in health care LLM evaluation and have emphasized that clearly defined evaluation dimensions and specific guidance for their application are important for improving the quality and interpretability of evaluations [<xref ref-type="bibr" rid="ref15">15</xref>]. Therefore, future studies need to develop a standardized indicator framework that clearly specifies the conceptual definitions and scope of application of each evaluation indicator and apply it consistently across studies.</p>
      </sec>
      <sec>
        <title>Diversity of Evaluator Composition and Misalignment Between Evaluators and Evaluation Indicators</title>
        <p>Across both domains, LLM reliability evaluation in health care has commonly relied on relatively small evaluator panels centered on clinicians. When evaluator panels are small, the influence of individual judgments on the overall findings is amplified, which may limit the stability and reproducibility of the results [<xref ref-type="bibr" rid="ref113">113</xref>]. In the clinical domain, most evaluations were conducted by clinicians from a single specialty. Although this may ensure the expertise required for clinical judgment, it may also be associated with a risk of bias toward the perspective of a particular specialty, given that real-world clinical decision-making is inherently multidisciplinary and involves collaboration across professions [<xref ref-type="bibr" rid="ref114">114</xref>]. In contrast, in the public health domain, mixed evaluator compositions involving nonclinicians, such as researchers, patients, and members of the general public, were more frequently observed. This may reflect efforts to incorporate the perspectives of diverse stakeholders.</p>
        <p>However, regardless of domain, some studies showed suboptimal alignment between evaluator expertise and evaluation indicators [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]. For example, clinicians assessed user-centered indicators, such as comprehensibility, empathy, or patient appropriateness [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]. Such misalignment may increase interpretive discrepancies among evaluators and potentially undermine the consistency and validity of measurement [<xref ref-type="bibr" rid="ref115">115</xref>]. Existing health care AI governance and evaluation frameworks have emphasized the importance of incorporating diverse stakeholder perspectives and multidisciplinary expertise into evaluation processes [<xref ref-type="bibr" rid="ref101">101</xref>,<xref ref-type="bibr" rid="ref116">116</xref>,<xref ref-type="bibr" rid="ref117">117</xref>]. However, our findings suggest that evaluator diversity alone may be insufficient, and that appropriate alignment between evaluator expertise and evaluation indicators may represent an additional methodological consideration for ensuring high-quality reliability assessment.</p>
      </sec>
      <sec>
        <title>The Need for Clear Judgment Criteria and Evaluator Guidance</title>
        <p>Likert scales and researcher-defined rubrics were the predominant evaluation approaches in both the clinical and public health domains. Likert scales facilitate comparisons across responses by quantifying subjective judgments, whereas researcher-defined rubrics allow detailed criteria to be tailored to the study purpose and the characteristics of the evaluation task [<xref ref-type="bibr" rid="ref118">118</xref>,<xref ref-type="bibr" rid="ref119">119</xref>]. However, some studies in this review identified differences in the interpretation of the same criteria, difficulty distinguishing between adjacent rating categories, and variation in assessments due to insufficient evaluator training as methodological limitations [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref85">85</xref>]. In particular, when the meaning and boundaries of rating categories are unclear, evaluators may assign scores based on criteria shaped by their own experience [<xref ref-type="bibr" rid="ref120">120</xref>,<xref ref-type="bibr" rid="ref121">121</xref>], which may hinder the accumulation of credible evidence in health care LLM evaluation [<xref ref-type="bibr" rid="ref22">22</xref>]. In health information evaluation, standardized tools have been used to systematically assess information based on clearly defined evaluation domains and scoring criteria [<xref ref-type="bibr" rid="ref122">122</xref>,<xref ref-type="bibr" rid="ref123">123</xref>]. Future LLM reliability evaluations need to consider drawing on these evaluation approaches to break down broad evaluation concepts into specific judgment items and clearly define the meaning of each score level along with representative response examples. Evaluator training and practice evaluations need to be conducted before the formal evaluation so that evaluators can develop a shared understanding of the criteria, clearly distinguish between categories, and apply common judgment principles [<xref ref-type="bibr" rid="ref124">124</xref>,<xref ref-type="bibr" rid="ref125">125</xref>]. Clear judgment criteria and evaluator guidance may reduce unnecessary interpretive variation arising during the evaluation process and improve interrater agreement, thereby contributing to the methodological rigor and reproducibility of health care LLM reliability evaluations [<xref ref-type="bibr" rid="ref126">126</xref>,<xref ref-type="bibr" rid="ref127">127</xref>].</p>
      </sec>
      <sec>
        <title>Methodological Limitations of Human Evaluation and Future Considerations for LLM Reliability in Health Care</title>
        <p>The methodological limitations of human evaluation identified in this review included evaluator subjectivity, the absence of standardized evaluation indicators, and limited evaluation scope and settings. In particular, human evaluation results may be influenced by subjective factors such as evaluators’ knowledge and experience and their interpretation of evaluation criteria [<xref ref-type="bibr" rid="ref128">128</xref>]. In addition, differences in evaluation design, including the range of questions and cases, interaction formats, evaluation scales, and criteria, may also affect evaluation results [<xref ref-type="bibr" rid="ref129">129</xref>]. These factors make it difficult to compare findings across studies and to accumulate consistent evidence on reliability [<xref ref-type="bibr" rid="ref130">130</xref>]. Nevertheless, human evaluation serves as an important complementary approach to benchmark-based evaluation because it can assess the context and qualitative characteristics of responses that are difficult to capture using quantitative evaluation metrics alone [<xref ref-type="bibr" rid="ref126">126</xref>,<xref ref-type="bibr" rid="ref131">131</xref>,<xref ref-type="bibr" rid="ref132">132</xref>]. Therefore, as emphasized in recent research on human evaluation of medical AI and LLMs, future human evaluations need to establish standardized frameworks that systematize evaluation indicators, criteria, and procedures while also considering evaluation designs that reflect the diverse situations and interactions of real-world health care settings [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref101">101</xref>]. Furthermore, as health care AI evolves toward agentic systems that integrate external tools, multistep reasoning, and autonomous decision-making capabilities, the scope of reliability evaluation may need to expand beyond the assessment of final responses alone [<xref ref-type="bibr" rid="ref133">133</xref>-<xref ref-type="bibr" rid="ref135">135</xref>]. Indeed, one previous study suggested that evaluating agentic AI may require additional evaluation dimensions, including planning, action execution, and error recovery, beyond those traditionally used for standalone LLMs [<xref ref-type="bibr" rid="ref136">136</xref>]. The common evaluation dimensions and methodological considerations identified in this review provide a useful foundation for distinguishing reliability aspects that can be assessed using existing LLM-centered criteria from those requiring evaluation dimensions specific to agentic AI systems, and may contribute to the development of reliability indicators and evaluation frameworks for agentic AI in health care.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>The findings of this review should be interpreted in light of several limitations. First, although preprints were included to capture recent developments in a rapidly evolving field, some findings may be less stable because these studies had not undergone formal peer review. Second, this review was conducted without formal protocol registration, which may limit the transparency and reproducibility of the review process. Third, some included studies did not provide sufficiently clear definitions of evaluation indicators or approaches, requiring researcher judgment during data charting, categorization, and interpretation. Although efforts were made to apply classification criteria consistently, some degree of subjectivity may have been involved in distinguishing between the clinical and public health domains and categorizing evaluation indicators, which may have influenced cross-study comparisons and interpretation of findings. Finally, studies included in this review were relatively concentrated in the public health domain, which may limit the extent to which characteristics of the clinical domain were represented. Accordingly, findings related to evaluation characteristics in the clinical domain should be interpreted with some caution. Nevertheless, to our knowledge, this is the first review to systematically examine the methodological structure of human evaluation–based LLM reliability assessment in health care, including evaluation indicators, evaluator composition, and evaluation approaches across clinical and public health domains.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>LLM reliability in health care is a context-dependent concept that cannot be adequately assessed using a single universal standard. Specifically, expert-centered clinical validity and logical rigor are key aspects of reliability in the clinical domain, whereas user-centered understandability, practical use, and safe use are key aspects in the public health domain. Nevertheless, common methodological challenges related to the design and conduct of human evaluations were identified across both domains. These issues may impede cross-study comparability and the systematic accumulation of reliability evidence. Therefore, it is important to establish a standardized reliability evaluation framework that reflects the characteristics and contexts of health care. Furthermore, as LLMs evolve into agentic AI systems, the scope of reliability evaluation needs to extend beyond final outputs to encompass reasoning processes and action selection. The evaluation dimensions and methodological considerations identified in this review are expected to provide a foundation for developing future reliability evaluation frameworks for LLMs and agentic AI in health care.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews) checklist.</p>
        <media xlink:href="jmir_v28i1e98184_app1.docx" xlink:title="DOCX File , 86 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Search strategies for electronic databases.</p>
        <media xlink:href="jmir_v28i1e98184_app2.doc" xlink:title="DOC File , 40 KB"/>
      </supplementary-material>
      <supplementary-material id="app3">
        <label>Multimedia Appendix 3</label>
        <p>Data chart of the included studies.</p>
        <media xlink:href="jmir_v28i1e98184_app3.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 61 KB"/>
      </supplementary-material>
      <supplementary-material id="app4">
        <label>Multimedia Appendix 4</label>
        <p>Studies contributing to each reported limitation category.</p>
        <media xlink:href="jmir_v28i1e98184_app4.docx" xlink:title="DOCX File , 38 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">BLEU</term>
          <def>
            <p>bilingual evaluation understudy</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">GQS</term>
          <def>
            <p>Global Quality Scale</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">PEMAT</term>
          <def>
            <p>Patient Education Materials Assessment Tool</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">PRISMA-ScR</term>
          <def>
            <p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">PRISMA-S</term>
          <def>
            <p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Searching</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors declare the use of generative AI (GAI) in the research and manuscript preparation process. According to the GAIDeT (Generative AI Delegation Taxonomy; 2025), GAI tools were used under full human supervision for the evaluation of research novelty and proofreading and editing. The GAI tools used were ChatGPT (OpenAI), including GPT-4.5 and GPT-5.5, and Claude Sonnet 4. Responsibility for the content and conclusions of the final manuscript lies entirely with the authors. GAI tools are not listed as authors and do not bear responsibility for the final outcomes. The declaration was submitted by EY.</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>The authors disclosed receipt of the following financial support for the research, authorship, and/or publication of this article: This work was supported by the National Research Foundation of Korea (NRF) grant funded by the Korean government (MSIT) (RS-2024-00350688).</p>
      </sec>
    </notes>
    <notes>
      <sec>
        <title>Data Availability</title>
        <p>The datasets used and/or analyzed during this study are available from the corresponding author upon reasonable request.</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Conceptualization: EY, HW</p>
        <p>Methodology: EY, SK, HW</p>
        <p>Data curation: EY, SK</p>
        <p>Formal analysis: EY</p>
        <p>Visualization: EY</p>
        <p>Writing – original draft: EY</p>
        <p>Writing – review and editing: EY, SK, HW</p>
        <p>Project administration: EY</p>
        <p>Supervision: HW</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>WX</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>A survey of large language models</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on March 31, 2023</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2303.18223</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vrdoljak</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Boban</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Vilović</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kumrić</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Božić</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>A review of large language models in medical education, clinical decision support, and healthcare administration</article-title>
          <source>Healthcare (Basel)</source>
          <year>2025</year>
          <month>03</month>
          <day>10</day>
          <volume>13</volume>
          <issue>6</issue>
          <fpage>603</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=healthcare13060603"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/healthcare13060603</pub-id>
          <pub-id pub-id-type="medline">40150453</pub-id>
          <pub-id pub-id-type="pii">healthcare13060603</pub-id>
          <pub-id pub-id-type="pmcid">PMC11942098</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>SF</given-names>
            </name>
            <name name-style="western">
              <surname>Alyakin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Seas</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Choi</surname>
              <given-names>JJ</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>JV</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>AL</given-names>
            </name>
            <name name-style="western">
              <surname>Warman</surname>
              <given-names>PI</given-names>
            </name>
            <name name-style="western">
              <surname>Bitolas</surname>
              <given-names>RT</given-names>
            </name>
            <name name-style="western">
              <surname>Steele</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Alber</surname>
              <given-names>DA</given-names>
            </name>
            <name name-style="western">
              <surname>Oermann</surname>
              <given-names>EK</given-names>
            </name>
          </person-group>
          <article-title>LLM-assisted systematic review of large language models in clinical medicine</article-title>
          <source>Nat Med</source>
          <year>2026</year>
          <month>03</month>
          <volume>32</volume>
          <issue>3</issue>
          <fpage>1152</fpage>
          <lpage>1159</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-026-04229-5</pub-id>
          <pub-id pub-id-type="medline">41776077</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-026-04229-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC13004689</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Topol</surname>
              <given-names>EJ</given-names>
            </name>
          </person-group>
          <article-title>High-performance medicine: the convergence of human and artificial intelligence</article-title>
          <source>Nat Med</source>
          <year>2019</year>
          <month>01</month>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>44</fpage>
          <lpage>56</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-018-0300-7</pub-id>
          <pub-id pub-id-type="medline">30617339</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-018-0300-7</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Zhong</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Feng</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Feng</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Qin</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>A survey on hallucination in large language models: principles, taxonomy, challenges, and open questions</article-title>
          <source>ACM Trans. Inf. Syst</source>
          <year>2025</year>
          <volume>43</volume>
          <issue>2</issue>
          <fpage>1</fpage>
          <lpage>55</lpage>
          <pub-id pub-id-type="doi">10.1145/3703155</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Maity</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Saikia</surname>
              <given-names>MJ</given-names>
            </name>
          </person-group>
          <article-title>Large language models in healthcare and medical applications: a review</article-title>
          <source>Bioengineering (Basel)</source>
          <year>2025</year>
          <month>06</month>
          <day>10</day>
          <volume>12</volume>
          <issue>6</issue>
          <fpage>631</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=bioengineering12060631"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/bioengineering12060631</pub-id>
          <pub-id pub-id-type="medline">40564447</pub-id>
          <pub-id pub-id-type="pii">bioengineering12060631</pub-id>
          <pub-id pub-id-type="pmcid">PMC12189880</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>van Kessel</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Anderson</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>McMillan</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Matthews</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Rust</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Pearcy</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Nasir</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Mossialos</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Omission and hallucination prevalence of clinical guidelines in diagnostic large language model outputs</article-title>
          <source>BMJ Health Care Inform</source>
          <year>2026</year>
          <month>04</month>
          <day>24</day>
          <volume>33</volume>
          <issue>1</issue>
          <fpage>e101959</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://informatics.bmj.com/lookup/pmidlookup?view=long&#38;pmid=42031418"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmjhci-2025-101959</pub-id>
          <pub-id pub-id-type="medline">42031418</pub-id>
          <pub-id pub-id-type="pii">bmjhci-2025-101959</pub-id>
          <pub-id pub-id-type="pmcid">PMC13110572</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bean</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>RE</given-names>
            </name>
            <name name-style="western">
              <surname>Parsons</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Kirk</surname>
              <given-names>HR</given-names>
            </name>
            <name name-style="western">
              <surname>Ciro</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Mosquera-Gómez</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Hincapié</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Ekanayaka</surname>
              <given-names>AS</given-names>
            </name>
            <name name-style="western">
              <surname>Tarassenko</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Rocher</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdi</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Reliability of LLMs as medical assistants for the general public: a randomized preregistered study</article-title>
          <source>Nat Med</source>
          <year>2026</year>
          <month>02</month>
          <volume>32</volume>
          <issue>2</issue>
          <fpage>609</fpage>
          <lpage>615</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-025-04074-y</pub-id>
          <pub-id pub-id-type="medline">41663592</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-025-04074-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC12920132</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Draelos</surname>
              <given-names>RL</given-names>
            </name>
            <name name-style="western">
              <surname>Afreen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Blasko</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Brazile</surname>
              <given-names>TL</given-names>
            </name>
            <name name-style="western">
              <surname>Chase</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Desai</surname>
              <given-names>DP</given-names>
            </name>
            <name name-style="western">
              <surname>Evert</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Gardner</surname>
              <given-names>HL</given-names>
            </name>
            <name name-style="western">
              <surname>Herrmann</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>House</surname>
              <given-names>AV</given-names>
            </name>
            <name name-style="western">
              <surname>Kass</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kavan</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Khemani</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Koire</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>McDonald</surname>
              <given-names>LM</given-names>
            </name>
            <name name-style="western">
              <surname>Rabeeah</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Large language models provide unsafe answers to patient-posed medical questions</article-title>
          <source>NPJ Digit Med</source>
          <year>2026</year>
          <month>02</month>
          <day>13</day>
          <volume>9</volume>
          <issue>1</issue>
          <fpage>241</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-026-02428-5"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-026-02428-5</pub-id>
          <pub-id pub-id-type="medline">41688533</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-026-02428-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC13013898</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Asgari</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Montaña-Brown</surname>
              <given-names>Nina</given-names>
            </name>
            <name name-style="western">
              <surname>Dubois</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Khalil</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Balloch</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yeung</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Pimenta</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>05</month>
          <day>13</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>274</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01670-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01670-7</pub-id>
          <pub-id pub-id-type="medline">40360677</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01670-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC12075489</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rajpurkar</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Banerjee</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Topol</surname>
              <given-names>EJ</given-names>
            </name>
          </person-group>
          <article-title>AI in health and medicine</article-title>
          <source>Nat Med</source>
          <year>2022</year>
          <volume>28</volume>
          <issue>1</issue>
          <fpage>31</fpage>
          <lpage>38</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-021-01614-0</pub-id>
          <pub-id pub-id-type="medline">35058619</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-021-01614-0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Royer</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Menze</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Sekuboyina</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Multimedeval: a benchmark and a toolkit for evaluating medical vision-language models</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on February 14, 2024</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2402.09262</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tanwani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Seneviratne</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gamble</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Babiker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schärli</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhery</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tomasev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>08</month>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>180</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tilton</surname>
              <given-names>AK</given-names>
            </name>
            <name name-style="western">
              <surname>Caplan</surname>
              <given-names>BE</given-names>
            </name>
            <name name-style="western">
              <surname>Cole</surname>
              <given-names>BJ</given-names>
            </name>
          </person-group>
          <article-title>Generative AI in consumer health: leveraging large language models for health literacy and clinical safety with a digital health framework</article-title>
          <source>Front Digit Health</source>
          <year>2025</year>
          <volume>7</volume>
          <fpage>1616488</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/fdgth.2025.1616488"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fdgth.2025.1616488</pub-id>
          <pub-id pub-id-type="medline">40933812</pub-id>
          <pub-id pub-id-type="pmcid">PMC12417475</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bedi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Orr-Ewing</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Dash</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Koyejo</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Callahan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Fries</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Wornow</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Swaminathan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lehmann</surname>
              <given-names>LS</given-names>
            </name>
            <name name-style="western">
              <surname>Hong</surname>
              <given-names>HJ</given-names>
            </name>
            <name name-style="western">
              <surname>Kashyap</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Chaurasia</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NR</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tazbaz</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Milstein</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pfeffer</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NH</given-names>
            </name>
          </person-group>
          <article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title>
          <source>JAMA</source>
          <year>2025</year>
          <month>01</month>
          <day>28</day>
          <volume>333</volume>
          <issue>4</issue>
          <fpage>319</fpage>
          <lpage>328</lpage>
          <pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id>
          <pub-id pub-id-type="medline">39405325</pub-id>
          <pub-id pub-id-type="pii">2825147</pub-id>
          <pub-id pub-id-type="pmcid">PMC11480901</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Xiang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Shi</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Evaluating large language models and agents in healthcare: key challenges in clinical applications</article-title>
          <source>Intelligent Med</source>
          <year>2025</year>
          <month>05</month>
          <volume>5</volume>
          <issue>2</issue>
          <fpage>151</fpage>
          <lpage>163</lpage>
          <pub-id pub-id-type="doi">10.1016/j.imed.2025.03.002</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ashqar</surname>
              <given-names>HI</given-names>
            </name>
          </person-group>
          <article-title>A critical review of benchmarking LLMs for real-world applications: trends and limitations</article-title>
          <year>2025</year>
          <conf-name>Sixteenth International Conference on Ubiquitous and Future Networks (ICUFN)</conf-name>
          <conf-date>July 8-11, 2025</conf-date>
          <conf-loc>Lisbon, Portugal</conf-loc>
          <pub-id pub-id-type="doi">10.1109/icufn65838.2025.11169931</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bommasani</surname>
              <given-names>Rishi</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>Percy</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>Tony</given-names>
            </name>
          </person-group>
          <article-title>Holistic evaluation of language models</article-title>
          <source>Ann N Y Acad Sci</source>
          <year>2023</year>
          <month>07</month>
          <volume>1525</volume>
          <issue>1</issue>
          <fpage>140</fpage>
          <lpage>146</lpage>
          <pub-id pub-id-type="doi">10.1111/nyas.15007</pub-id>
          <pub-id pub-id-type="medline">37230490</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Budler</surname>
              <given-names>LC</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Topaz</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Tam</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Bian</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Stiglic</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>A brief review on benchmarking for large language models evaluation in healthcare</article-title>
          <source>WIREs Data Min &#38; Knowl</source>
          <year>2025</year>
          <month>04</month>
          <day>09</day>
          <volume>15</volume>
          <issue>2</issue>
          <fpage>e70010</fpage>
          <pub-id pub-id-type="doi">10.1002/widm.70010</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Gong</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Gu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Lian</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Mao</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ji</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Niu</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Lang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Meng</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>A novel evaluation benchmark for medical LLMs illuminating safety and effectiveness in clinical domains</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>12</month>
          <day>26</day>
          <volume>9</volume>
          <issue>1</issue>
          <fpage>91</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-025-02277-8</pub-id>
          <pub-id pub-id-type="medline">41454006</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-02277-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC12855988</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yi</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>PS</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>A survey on evaluation of large language models</article-title>
          <source>ACM Trans Intell Syst Technol</source>
          <year>2024</year>
          <volume>15</volume>
          <issue>3</issue>
          <fpage>1</fpage>
          <lpage>45</lpage>
          <pub-id pub-id-type="doi">10.1145/3641289</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Awasthi</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Bhattad</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Ramachandran</surname>
              <given-names>SP</given-names>
            </name>
            <name name-style="western">
              <surname>Mishra</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Khanna</surname>
              <given-names>AK</given-names>
            </name>
            <name name-style="western">
              <surname>Cywinski</surname>
              <given-names>JB</given-names>
            </name>
            <name name-style="western">
              <surname>Maheshwari</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Mahapatra</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>DiRosa</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Cohen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Arshad</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Atreja</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Alshukaili</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Vohra</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Papay</surname>
              <given-names>FA</given-names>
            </name>
            <name name-style="western">
              <surname>Atreja</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kashyap</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Mathur</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Human evaluation of large language models in healthcare: gaps, challenges, and the need for standardization</article-title>
          <source>Npj Health Syst</source>
          <year>2025</year>
          <month>11</month>
          <day>03</day>
          <volume>2</volume>
          <issue>1</issue>
          <fpage>40</fpage>
          <pub-id pub-id-type="doi">10.1038/s44401-025-00043-2</pub-id>
          <pub-id pub-id-type="medline">42527491</pub-id>
          <pub-id pub-id-type="pii">10.1038/s44401-025-00043-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC13354174</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tricco</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Lillie</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Zarin</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>O'Brien</surname>
              <given-names>KK</given-names>
            </name>
            <name name-style="western">
              <surname>Colquhoun</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Levac</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Moher</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Peters</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Horsley</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Weeks</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Hempel</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Akl</surname>
              <given-names>EA</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>McGowan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Stewart</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Hartling</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Aldcroft</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wilson</surname>
              <given-names>MG</given-names>
            </name>
            <name name-style="western">
              <surname>Garritty</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lewin</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Godfrey</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Macdonald</surname>
              <given-names>MT</given-names>
            </name>
            <name name-style="western">
              <surname>Langlois</surname>
              <given-names>EV</given-names>
            </name>
            <name name-style="western">
              <surname>Soares-Weiser</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Moriarty</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Clifford</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Tunçalp</surname>
              <given-names>Özge</given-names>
            </name>
            <name name-style="western">
              <surname>Straus</surname>
              <given-names>SE</given-names>
            </name>
          </person-group>
          <article-title>PRISMA Extension for Scoping Reviews (PRISMA-ScR): checklist and explanation</article-title>
          <source>Ann Intern Med</source>
          <year>2018</year>
          <month>10</month>
          <day>02</day>
          <volume>169</volume>
          <issue>7</issue>
          <fpage>467</fpage>
          <lpage>473</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.acpjournals.org/doi/10.7326/M18-0850?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.7326/M18-0850</pub-id>
          <pub-id pub-id-type="medline">30178033</pub-id>
          <pub-id pub-id-type="pii">2700389</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rethlefsen</surname>
              <given-names>ML</given-names>
            </name>
            <name name-style="western">
              <surname>Kirtley</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Waffenschmidt</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ayala</surname>
              <given-names>AP</given-names>
            </name>
            <name name-style="western">
              <surname>Moher</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Page</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Koffel</surname>
              <given-names>JB</given-names>
            </name>
            <collab>PRISMA-S Group</collab>
          </person-group>
          <article-title>PRISMA-S: an extension to the PRISMA Statement for Reporting Literature Searches in Systematic Reviews</article-title>
          <source>Syst Rev</source>
          <year>2021</year>
          <month>01</month>
          <day>26</day>
          <volume>10</volume>
          <issue>1</issue>
          <fpage>39</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://systematicreviewsjournal.biomedcentral.com/articles/10.1186/s13643-020-01542-z"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s13643-020-01542-z</pub-id>
          <pub-id pub-id-type="medline">33499930</pub-id>
          <pub-id pub-id-type="pii">10.1186/s13643-020-01542-z</pub-id>
          <pub-id pub-id-type="pmcid">PMC7839230</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Haddaway</surname>
              <given-names>NR</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Coughlin</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kirk</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>The role of Google scholar in evidence reviews and its applicability to grey literature searching</article-title>
          <source>PLoS One</source>
          <year>2015</year>
          <volume>10</volume>
          <issue>9</issue>
          <fpage>e0138237</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pone.0138237"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0138237</pub-id>
          <pub-id pub-id-type="medline">26379270</pub-id>
          <pub-id pub-id-type="pii">PONE-D-15-27398</pub-id>
          <pub-id pub-id-type="pmcid">PMC4574933</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hua</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Na</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Fang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Clifton</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Torous</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>A scoping review of large language models for generative tasks in mental health care</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>04</month>
          <day>30</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>230</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01611-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01611-4</pub-id>
          <pub-id pub-id-type="medline">40307331</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01611-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC12043943</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Moulaei</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Yadegari</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Baharestani</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Farzanbakhsh</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sabet</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Reza Afrash</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Generative artificial intelligence in healthcare: a scoping review on benefits, challenges and applications</article-title>
          <source>Int J Med Inform</source>
          <year>2024</year>
          <month>08</month>
          <volume>188</volume>
          <fpage>105474</fpage>
          <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2024.105474</pub-id>
          <pub-id pub-id-type="medline">38733640</pub-id>
          <pub-id pub-id-type="pii">S1386-5056(24)00137-0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Peters</surname>
              <given-names>MDJ</given-names>
            </name>
            <name name-style="western">
              <surname>Marnie</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Tricco</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Pollock</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Munn</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Alexander</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>McInerney</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Godfrey</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Khalil</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Updated methodological guidance for the conduct of scoping reviews</article-title>
          <source>JBI Evid Synth</source>
          <year>2020</year>
          <month>10</month>
          <volume>18</volume>
          <issue>10</issue>
          <fpage>2119</fpage>
          <lpage>2126</lpage>
          <pub-id pub-id-type="doi">10.11124/JBIES-20-00167</pub-id>
          <pub-id pub-id-type="medline">33038124</pub-id>
          <pub-id pub-id-type="pii">02174543-202010000-00004</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ouzzani</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hammady</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Fedorowicz</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Elmagarmid</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Rayyan-a web and mobile app for systematic reviews</article-title>
          <source>Syst Rev</source>
          <year>2016</year>
          <month>12</month>
          <day>05</day>
          <volume>5</volume>
          <issue>1</issue>
          <fpage>210</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://systematicreviewsjournal.biomedcentral.com/articles/10.1186/s13643-016-0384-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s13643-016-0384-4</pub-id>
          <pub-id pub-id-type="medline">27919275</pub-id>
          <pub-id pub-id-type="pii">10.1186/s13643-016-0384-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC5139140</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kotzur</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Parker</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Peterson</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Sager</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Rose</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Corley</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Brady</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of a large language model's ability to assist in an orthopedic hand clinic</article-title>
          <source>Hand (N Y)</source>
          <year>2025</year>
          <month>09</month>
          <volume>20</volume>
          <issue>6</issue>
          <fpage>900</fpage>
          <lpage>909</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/15589447241257643?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/15589447241257643</pub-id>
          <pub-id pub-id-type="medline">38907651</pub-id>
          <pub-id pub-id-type="pmcid">PMC11571334</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kuerbanjiang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Jiamaliding</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yi</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Performance evaluation of large language models in cervical cancer management based on a standardized questionnaire: comparative study</article-title>
          <source>J Med Internet Res</source>
          <year>2025</year>
          <month>02</month>
          <day>05</day>
          <volume>27</volume>
          <fpage>e63626</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2025//e63626/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/63626</pub-id>
          <pub-id pub-id-type="medline">39908540</pub-id>
          <pub-id pub-id-type="pii">v27i1e63626</pub-id>
          <pub-id pub-id-type="pmcid">PMC11840365</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ying</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ding</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Screening/diagnosis of pediatric endocrine disorders through the artificial intelligence model in different language settings</article-title>
          <source>Eur J Pediatr</source>
          <year>2024</year>
          <month>06</month>
          <volume>183</volume>
          <issue>6</issue>
          <fpage>2655</fpage>
          <lpage>2661</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38502320"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s00431-024-05527-1</pub-id>
          <pub-id pub-id-type="medline">38502320</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00431-024-05527-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC11098926</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Leypold</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Lingens</surname>
              <given-names>LF</given-names>
            </name>
            <name name-style="western">
              <surname>Beier</surname>
              <given-names>JP</given-names>
            </name>
            <name name-style="western">
              <surname>Boos</surname>
              <given-names>AM</given-names>
            </name>
          </person-group>
          <article-title>Integrating AI in lipedema management: assessing the efficacy of GPT-4 as a consultation assistant</article-title>
          <source>Life (Basel)</source>
          <year>2024</year>
          <month>05</month>
          <day>20</day>
          <volume>14</volume>
          <issue>5</issue>
          <fpage>646</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=life14050646"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/life14050646</pub-id>
          <pub-id pub-id-type="medline">38792666</pub-id>
          <pub-id pub-id-type="pii">life14050646</pub-id>
          <pub-id pub-id-type="pmcid">PMC11122530</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Leypold</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Bahm</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Beier</surname>
              <given-names>JP</given-names>
            </name>
            <name name-style="western">
              <surname>Guillaume</surname>
              <given-names>VG</given-names>
            </name>
            <name name-style="western">
              <surname>Ammo</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Lauer</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Kolbenschlag</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Schäfer</surname>
              <given-names>Benedikt</given-names>
            </name>
          </person-group>
          <article-title>Evaluating ChatGPT o1's capabilities in peripheral nerve surgery: advancing artificial intelligence in clinical practice</article-title>
          <source>World Neurosurg</source>
          <year>2025</year>
          <month>04</month>
          <volume>196</volume>
          <fpage>123753</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1878-8750(25)00109-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.wneu.2025.123753</pub-id>
          <pub-id pub-id-type="medline">39924104</pub-id>
          <pub-id pub-id-type="pii">S1878-8750(25)00109-3</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cankurtaran</surname>
              <given-names>RE</given-names>
            </name>
            <name name-style="western">
              <surname>Polat</surname>
              <given-names>YH</given-names>
            </name>
            <name name-style="western">
              <surname>Aydemir</surname>
              <given-names>NG</given-names>
            </name>
            <name name-style="western">
              <surname>Umay</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Yurekli</surname>
              <given-names>OT</given-names>
            </name>
          </person-group>
          <article-title>Reliability and usefulness of ChatGPT for inflammatory bowel diseases: an analysis for patients and healthcare professionals</article-title>
          <source>Cureus</source>
          <year>2023</year>
          <month>10</month>
          <volume>15</volume>
          <issue>10</issue>
          <fpage>e46736</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38022227"/>
          </comment>
          <pub-id pub-id-type="doi">10.7759/cureus.46736</pub-id>
          <pub-id pub-id-type="medline">38022227</pub-id>
          <pub-id pub-id-type="pmcid">PMC10630704</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Goodman</surname>
              <given-names>RS</given-names>
            </name>
            <name name-style="western">
              <surname>Patrinely</surname>
              <given-names>JR</given-names>
            </name>
            <name name-style="western">
              <surname>Stone</surname>
              <given-names>CA</given-names>
            </name>
            <name name-style="western">
              <surname>Zimmerman</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Donald</surname>
              <given-names>RR</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Berkowitz</surname>
              <given-names>ST</given-names>
            </name>
            <name name-style="western">
              <surname>Finn</surname>
              <given-names>AP</given-names>
            </name>
            <name name-style="western">
              <surname>Jahangir</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Scoville</surname>
              <given-names>EA</given-names>
            </name>
            <name name-style="western">
              <surname>Reese</surname>
              <given-names>TS</given-names>
            </name>
            <name name-style="western">
              <surname>Friedman</surname>
              <given-names>DL</given-names>
            </name>
            <name name-style="western">
              <surname>Bastarache</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>van der Heijden</surname>
              <given-names>YF</given-names>
            </name>
            <name name-style="western">
              <surname>Wright</surname>
              <given-names>JJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Carter</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Alexander</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Choe</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Chastain</surname>
              <given-names>CA</given-names>
            </name>
            <name name-style="western">
              <surname>Zic</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Horst</surname>
              <given-names>SN</given-names>
            </name>
            <name name-style="western">
              <surname>Turker</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Agarwal</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Osmundson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Idrees</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Kiernan</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Padmanabhan</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Bailey</surname>
              <given-names>CE</given-names>
            </name>
            <name name-style="western">
              <surname>Schlegel</surname>
              <given-names>CE</given-names>
            </name>
            <name name-style="western">
              <surname>Chambless</surname>
              <given-names>LB</given-names>
            </name>
            <name name-style="western">
              <surname>Gibson</surname>
              <given-names>MK</given-names>
            </name>
            <name name-style="western">
              <surname>Osterman</surname>
              <given-names>TJ</given-names>
            </name>
            <name name-style="western">
              <surname>Wheless</surname>
              <given-names>LE</given-names>
            </name>
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>DB</given-names>
            </name>
          </person-group>
          <article-title>Accuracy and reliability of chatbot responses to physician questions</article-title>
          <source>JAMA Netw Open</source>
          <year>2023</year>
          <month>10</month>
          <day>02</day>
          <volume>6</volume>
          <issue>10</issue>
          <fpage>e2336483</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37782499"/>
          </comment>
          <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.36483</pub-id>
          <pub-id pub-id-type="medline">37782499</pub-id>
          <pub-id pub-id-type="pii">2809975</pub-id>
          <pub-id pub-id-type="pmcid">PMC10546234</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Draschl</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Hauer</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Fischerauer</surname>
              <given-names>SF</given-names>
            </name>
            <name name-style="western">
              <surname>Kogler</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Leitner</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Andreou</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Leithner</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sadoghi</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Are ChatGPT's free-text responses on periprosthetic joint infections of the hip and knee reliable and useful?</article-title>
          <source>J Clin Med</source>
          <year>2023</year>
          <month>10</month>
          <day>20</day>
          <volume>12</volume>
          <issue>20</issue>
          <fpage>6655</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=jcm12206655"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/jcm12206655</pub-id>
          <pub-id pub-id-type="medline">37892793</pub-id>
          <pub-id pub-id-type="pii">jcm12206655</pub-id>
          <pub-id pub-id-type="pmcid">PMC10607052</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Leypold</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Schäfer</surname>
              <given-names>Benedikt</given-names>
            </name>
            <name name-style="western">
              <surname>Boos</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Beier</surname>
              <given-names>JP</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence-powered hand surgery consultation: GPT-4 as an assistant in a hand surgery outpatient clinic</article-title>
          <source>J Hand Surg Am</source>
          <year>2024</year>
          <month>11</month>
          <volume>49</volume>
          <issue>11</issue>
          <fpage>1078</fpage>
          <lpage>1088</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S0363-5023(24)00261-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jhsa.2024.06.002</pub-id>
          <pub-id pub-id-type="medline">39066762</pub-id>
          <pub-id pub-id-type="pii">S0363-5023(24)00261-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vaira</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Lechien</surname>
              <given-names>JR</given-names>
            </name>
            <name name-style="western">
              <surname>Abbate</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Allevi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Audino</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Beltramini</surname>
              <given-names>GA</given-names>
            </name>
            <name name-style="western">
              <surname>Bergonzani</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Bolzoni</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Committeri</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Crimi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Gabriele</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Lonardi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Maglitto</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Petrocelli</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pucci</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Saponaro</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Tel</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Vellone</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Chiesa-Estomba</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Boscolo-Rizzo</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Salzano</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>De Riu</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Accuracy of chatGPT-generated information on head and neck and oromaxillofacial surgery: a multicenter collaborative analysis</article-title>
          <source>Otolaryngol Head Neck Surg</source>
          <year>2024</year>
          <month>06</month>
          <volume>170</volume>
          <issue>6</issue>
          <fpage>1492</fpage>
          <lpage>1503</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://air.unimi.it/handle/2434/1024627"/>
          </comment>
          <pub-id pub-id-type="doi">10.1002/ohn.489</pub-id>
          <pub-id pub-id-type="medline">37595113</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref40">
        <label>40</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Makrygiannakis</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Giannakopoulos</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Kaklamanos</surname>
              <given-names>EG</given-names>
            </name>
          </person-group>
          <article-title>Evidence-based potential of generative artificial intelligence large language models in orthodontics: a comparative study of ChatGPT, Google Bard, and Microsoft Bing</article-title>
          <source>Eur J Orthod</source>
          <year>2025</year>
          <month>12</month>
          <day>16</day>
          <volume>48</volume>
          <issue>1</issue>
          <fpage>cjae017</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/ejo/article-lookup/doi/10.1093/ejo/cjae017"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/ejo/cjae017</pub-id>
          <pub-id pub-id-type="medline">38613510</pub-id>
          <pub-id pub-id-type="pii">7645326</pub-id>
          <pub-id pub-id-type="pmcid">PMC12810200</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref41">
        <label>41</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wilhelm</surname>
              <given-names>TI</given-names>
            </name>
            <name name-style="western">
              <surname>Roos</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kaczmarczyk</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Large language models for therapy recommendations across 3 clinical specialties: comparative study</article-title>
          <source>J Med Internet Res</source>
          <year>2023</year>
          <volume>25</volume>
          <fpage>e49324</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2023//e49324/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/49324</pub-id>
          <pub-id pub-id-type="medline">37902826</pub-id>
          <pub-id pub-id-type="pii">v25i1e49324</pub-id>
          <pub-id pub-id-type="pmcid">PMC10644179</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref42">
        <label>42</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Fraile Navarro</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Coiera</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Hambly</surname>
              <given-names>TW</given-names>
            </name>
            <name name-style="western">
              <surname>Triplett</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Asif</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Susanto</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhury</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Azcoaga Lorenzo</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dras</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Berkovsky</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Expert evaluation of large language models for clinical dialogue summarization</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <volume>15</volume>
          <issue>1</issue>
          <fpage>1195</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-024-84850-x"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-024-84850-x</pub-id>
          <pub-id pub-id-type="medline">39774141</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-024-84850-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC11707028</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref43">
        <label>43</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Abola</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Bargnes</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Tsivitis</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Rahman</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Schwartz</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bergese</surname>
              <given-names>SD</given-names>
            </name>
            <name name-style="western">
              <surname>Schabel</surname>
              <given-names>JE</given-names>
            </name>
          </person-group>
          <article-title>The utility of generative artificial intelligence Chatbot (ChatGPT) in generating teaching and learning material for anesthesiology residents</article-title>
          <source>Front Artif Intell</source>
          <year>2025</year>
          <volume>8</volume>
          <fpage>1582096</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/frai.2025.1582096"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/frai.2025.1582096</pub-id>
          <pub-id pub-id-type="medline">40469072</pub-id>
          <pub-id pub-id-type="pmcid">PMC12133725</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref44">
        <label>44</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rubinstein</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mohsin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Banerjee</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Mishra</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kwok</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Warner</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Cowan</surname>
              <given-names>AJ</given-names>
            </name>
          </person-group>
          <article-title>Summarizing clinical evidence utilizing large language models for cancer treatments: a blinded comparative analysis</article-title>
          <source>Front Digit Health</source>
          <year>2025</year>
          <volume>7</volume>
          <fpage>1569554</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/fdgth.2025.1569554"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fdgth.2025.1569554</pub-id>
          <pub-id pub-id-type="medline">40364850</pub-id>
          <pub-id pub-id-type="pmcid">PMC12069342</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref45">
        <label>45</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Balas</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Mandelcorn</surname>
              <given-names>ED</given-names>
            </name>
            <name name-style="western">
              <surname>Yan</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Ing</surname>
              <given-names>EB</given-names>
            </name>
            <name name-style="western">
              <surname>Crawford</surname>
              <given-names>SA</given-names>
            </name>
            <name name-style="western">
              <surname>Arjmand</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT and retinal disease: a cross-sectional study on AI comprehension of clinical guidelines</article-title>
          <source>Can J Ophthalmol</source>
          <year>2025</year>
          <volume>60</volume>
          <issue>1</issue>
          <fpage>e117</fpage>
          <lpage>e123</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S0008-4182(24)00175-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jcjo.2024.06.001</pub-id>
          <pub-id pub-id-type="medline">39097289</pub-id>
          <pub-id pub-id-type="pii">S0008-4182(24)00175-3</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref46">
        <label>46</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Alqudah</surname>
              <given-names>AA</given-names>
            </name>
            <name name-style="western">
              <surname>Aleshawi</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Baker</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Alnajjar</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Ayasrah</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Ta'ani</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Al Salkhadi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Aljawarneh</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Evaluating accuracy and reproducibility of ChatGPT responses to patient-based questions in ophthalmology: an observational study</article-title>
          <source>Medicine (Baltimore)</source>
          <year>2024</year>
          <volume>103</volume>
          <issue>32</issue>
          <fpage>e39120</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.ovid.com/10.1097/MD.0000000000039120"/>
          </comment>
          <pub-id pub-id-type="doi">10.1097/MD.0000000000039120</pub-id>
          <pub-id pub-id-type="medline">39121263</pub-id>
          <pub-id pub-id-type="pii">00005792-202408090-00020</pub-id>
          <pub-id pub-id-type="pmcid">PMC11315477</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref47">
        <label>47</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lima</surname>
              <given-names>HA</given-names>
            </name>
            <name name-style="western">
              <surname>Trocoli-Couto</surname>
              <given-names>PHFS</given-names>
            </name>
            <name name-style="western">
              <surname>Moazzam</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Rocha</surname>
              <given-names>LCD</given-names>
            </name>
            <name name-style="western">
              <surname>Pagano</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Martins</surname>
              <given-names>FF</given-names>
            </name>
            <name name-style="western">
              <surname>Brabo</surname>
              <given-names>LT</given-names>
            </name>
            <name name-style="western">
              <surname>Reis</surname>
              <given-names>ZSN</given-names>
            </name>
            <name name-style="western">
              <surname>Keder</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Begum</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mamede</surname>
              <given-names>MH</given-names>
            </name>
            <name name-style="western">
              <surname>Pawlik</surname>
              <given-names>TM</given-names>
            </name>
            <name name-style="western">
              <surname>Resende</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Quality assessment of large language models' output in maternal health</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <volume>15</volume>
          <issue>1</issue>
          <fpage>22474</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-025-03501-x"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-025-03501-x</pub-id>
          <pub-id pub-id-type="medline">40593918</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-025-03501-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC12215737</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref48">
        <label>48</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Hao</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Comparison of the performance of ChatGPT, claude and bard in support of myopia prevention and control</article-title>
          <source>J Multidiscip Healthc</source>
          <year>2024</year>
          <volume>17</volume>
          <fpage>3917</fpage>
          <lpage>3929</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.tandfonline.com/doi/10.2147/JMDH.S473680?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.2147/JMDH.S473680</pub-id>
          <pub-id pub-id-type="medline">39155977</pub-id>
          <pub-id pub-id-type="pii">473680</pub-id>
          <pub-id pub-id-type="pmcid">PMC11330241</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref49">
        <label>49</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Büker</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Mercan</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Readability, accuracy and appropriateness and quality of AI chatbot responses as a patient information source on root canal retreatment: a comparative assessment</article-title>
          <source>Int J Med Inform</source>
          <year>2025</year>
          <volume>201</volume>
          <fpage>105948</fpage>
          <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.105948</pub-id>
          <pub-id pub-id-type="medline">40288015</pub-id>
          <pub-id pub-id-type="pii">S1386-5056(25)00165-0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref50">
        <label>50</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yan</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Shao</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Assessing large language models as assistive tools in medical consultations for Kawasaki disease</article-title>
          <source>Front Artif Intell</source>
          <year>2025</year>
          <volume>8</volume>
          <fpage>1571503</fpage>
          <pub-id pub-id-type="doi">10.3389/frai.2025.1571503</pub-id>
          <pub-id pub-id-type="medline">40231209</pub-id>
          <pub-id pub-id-type="pmcid">PMC11994668</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref51">
        <label>51</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Roldan-Vasquez</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Mitri</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Bhasin</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Bharani</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Capasso</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Haslinger</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sharma</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>James</surname>
              <given-names>TA</given-names>
            </name>
          </person-group>
          <article-title>Reliability of artificial intelligence chatbot responses to frequently asked questions in breast surgical oncology</article-title>
          <source>J Surg Oncol</source>
          <year>2024</year>
          <volume>130</volume>
          <issue>2</issue>
          <fpage>188</fpage>
          <lpage>203</lpage>
          <pub-id pub-id-type="doi">10.1002/jso.27715</pub-id>
          <pub-id pub-id-type="medline">38837375</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref52">
        <label>52</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yau</surname>
              <given-names>JY</given-names>
            </name>
            <name name-style="western">
              <surname>Saadat</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hsu</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Murphy</surname>
              <given-names>LS</given-names>
            </name>
            <name name-style="western">
              <surname>Roh</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>Suchard</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tapia</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wiechmann</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Langdorf</surname>
              <given-names>MI</given-names>
            </name>
          </person-group>
          <article-title>Accuracy of prospective assessments of 4 large language model chatbot responses to patient questions about emergency care: experimental comparative study</article-title>
          <source>J Med Internet Res</source>
          <year>2024</year>
          <volume>26</volume>
          <fpage>e60291</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2024//e60291/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/60291</pub-id>
          <pub-id pub-id-type="medline">39496149</pub-id>
          <pub-id pub-id-type="pii">v26i1e60291</pub-id>
          <pub-id pub-id-type="pmcid">PMC11574488</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref53">
        <label>53</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sezgin</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Jackson</surname>
              <given-names>DI</given-names>
            </name>
            <name name-style="western">
              <surname>Kocaballi</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Bibart</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zupanec</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Landier</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Audino</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Ranalli</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Skeens</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Can large language models aid caregivers of pediatric cancer patients in information seeking? A cross-sectional investigation</article-title>
          <source>Cancer Med</source>
          <year>2025</year>
          <volume>14</volume>
          <issue>1</issue>
          <fpage>e70554</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://onlinelibrary.wiley.com/doi/10.1002/cam4.70554"/>
          </comment>
          <pub-id pub-id-type="doi">10.1002/cam4.70554</pub-id>
          <pub-id pub-id-type="medline">39776222</pub-id>
          <pub-id pub-id-type="pmcid">PMC11705392</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref54">
        <label>54</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Motegi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Shino</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kuwabara</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Takahashi</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Matsuyama</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Tada</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Hagiwara</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Chikamatsu</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Comparison of physician and large language model chatbot responses to online ear, nose, and throat inquiries</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <volume>15</volume>
          <issue>1</issue>
          <fpage>21346</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-025-06769-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-025-06769-1</pub-id>
          <pub-id pub-id-type="medline">40596359</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-025-06769-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC12215459</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref55">
        <label>55</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kamal</surname>
              <given-names>AH</given-names>
            </name>
          </person-group>
          <article-title>AI chatbots in pediatric orthopedics: how accurate are their answers to parents' questions on bowlegs and knock knees?</article-title>
          <source>Healthcare (Basel)</source>
          <year>2025</year>
          <volume>13</volume>
          <issue>11</issue>
          <fpage>1271</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=healthcare13111271"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/healthcare13111271</pub-id>
          <pub-id pub-id-type="medline">40508883</pub-id>
          <pub-id pub-id-type="pii">healthcare13111271</pub-id>
          <pub-id pub-id-type="pmcid">PMC12154324</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref56">
        <label>56</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Saad</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Moqeet</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Mansoor</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Khan</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sharif</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Khan</surname>
              <given-names>FU</given-names>
            </name>
            <name name-style="western">
              <surname>Naqvi</surname>
              <given-names>AH</given-names>
            </name>
            <name name-style="western">
              <surname>Ali</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the efficacy of artificial intelligence-driven chatbots in addressing queries on vernal conjunctivitis</article-title>
          <source>Cureus</source>
          <year>2025</year>
          <volume>17</volume>
          <issue>2</issue>
          <fpage>e79688</fpage>
          <pub-id pub-id-type="doi">10.7759/cureus.79688</pub-id>
          <pub-id pub-id-type="medline">40161163</pub-id>
          <pub-id pub-id-type="pmcid">PMC11951947</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref57">
        <label>57</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yaş</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Yapar</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Yapar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Özel</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Tokgöz</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Baymurat</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Şenköylü</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Assessing the role of large language models in adolescent idiopathic scoliosis care: a comparison between ChatGPT and Google Gemini</article-title>
          <source>Acta Orthop Traumatol Turc</source>
          <year>2025</year>
          <volume>59</volume>
          <issue>4</issue>
          <fpage>222</fpage>
          <lpage>229</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.5152/j.aott.2025.25279"/>
          </comment>
          <pub-id pub-id-type="doi">10.5152/j.aott.2025.25279</pub-id>
          <pub-id pub-id-type="medline">40728044</pub-id>
          <pub-id pub-id-type="pmcid">PMC12362497</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref58">
        <label>58</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hassona</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Alqaisi</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Al-Haddad</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Georgakopoulou</surname>
              <given-names>EA</given-names>
            </name>
            <name name-style="western">
              <surname>Malamos</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Alrashdan</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Sawair</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>How good is ChatGPT at answering patients' questions related to early detection of oral (mouth) cancer?</article-title>
          <source>Oral Surg Oral Med Oral Pathol Oral Radiol</source>
          <year>2024</year>
          <volume>138</volume>
          <issue>2</issue>
          <fpage>269</fpage>
          <lpage>278</lpage>
          <pub-id pub-id-type="doi">10.1016/j.oooo.2024.04.010</pub-id>
          <pub-id pub-id-type="medline">38714483</pub-id>
          <pub-id pub-id-type="pii">S2212-4403(24)00164-0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref59">
        <label>59</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sahin Ozdemir</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ozdemir</surname>
              <given-names>YE</given-names>
            </name>
          </person-group>
          <article-title>Comparison of the performances between ChatGPT and Gemini in answering questions on viral hepatitis</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <volume>15</volume>
          <issue>1</issue>
          <fpage>1712</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-024-83575-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-024-83575-1</pub-id>
          <pub-id pub-id-type="medline">39799203</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-024-83575-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC11724965</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref60">
        <label>60</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lang</surname>
              <given-names>SP</given-names>
            </name>
            <name name-style="western">
              <surname>Yoseph</surname>
              <given-names>ET</given-names>
            </name>
            <name name-style="western">
              <surname>Gonzalez-Suarez</surname>
              <given-names>AD</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Fatemi</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Wagner</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Maldaner</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Stienen</surname>
              <given-names>MN</given-names>
            </name>
            <name name-style="western">
              <surname>Zygourakis</surname>
              <given-names>CC</given-names>
            </name>
          </person-group>
          <article-title>Analyzing large language models' responses to common lumbar spine fusion surgery questions: a comparison between ChatGPT and bard</article-title>
          <source>Neurospine</source>
          <year>2024</year>
          <volume>21</volume>
          <issue>2</issue>
          <fpage>633</fpage>
          <lpage>641</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://e-neurospine.org/journal/view.php?doi=10.14245/ns.2448098.049"/>
          </comment>
          <pub-id pub-id-type="doi">10.14245/ns.2448098.049</pub-id>
          <pub-id pub-id-type="medline">38955533</pub-id>
          <pub-id pub-id-type="pii">ns.2448098.049</pub-id>
          <pub-id pub-id-type="pmcid">PMC11224745</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref61">
        <label>61</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Piao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Assessing the performance of large language models (LLMs) in answering medical questions regarding breast cancer in the Chinese context</article-title>
          <source>Digit Health</source>
          <year>2024</year>
          <volume>10</volume>
          <fpage>20552076241284771</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/20552076241284771?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/20552076241284771</pub-id>
          <pub-id pub-id-type="medline">39386109</pub-id>
          <pub-id pub-id-type="pii">10.1177_20552076241284771</pub-id>
          <pub-id pub-id-type="pmcid">PMC11462564</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref62">
        <label>62</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Uz</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Umay</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>"Dr ChatGPT": is it a reliable and useful source for common rheumatic diseases?</article-title>
          <source>Int J Rheum Dis</source>
          <year>2023</year>
          <volume>26</volume>
          <issue>7</issue>
          <fpage>1343</fpage>
          <lpage>1349</lpage>
          <pub-id pub-id-type="doi">10.1111/1756-185X.14749</pub-id>
          <pub-id pub-id-type="medline">37218530</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref63">
        <label>63</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mahedia</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rohrich</surname>
              <given-names>RN</given-names>
            </name>
            <name name-style="western">
              <surname>Sadiq</surname>
              <given-names>KO</given-names>
            </name>
            <name name-style="western">
              <surname>Bailey</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Harrison</surname>
              <given-names>LM</given-names>
            </name>
            <name name-style="western">
              <surname>Hallac</surname>
              <given-names>RR</given-names>
            </name>
          </person-group>
          <article-title>Exploring the utility of ChatGPT in cleft lip repair education</article-title>
          <source>J Clin Med</source>
          <year>2025</year>
          <volume>14</volume>
          <issue>3</issue>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=jcm14030993"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/jcm14030993</pub-id>
          <pub-id pub-id-type="medline">39941663</pub-id>
          <pub-id pub-id-type="pii">jcm14030993</pub-id>
          <pub-id pub-id-type="pmcid">PMC11818196</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref64">
        <label>64</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lan</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Teng</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Comparative evaluation of the accuracy and reliability of ChatGPT versions in providing information on infection</article-title>
          <source>Front Public Health</source>
          <year>2025</year>
          <volume>13</volume>
          <fpage>1566982</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/fpubh.2025.1566982"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fpubh.2025.1566982</pub-id>
          <pub-id pub-id-type="medline">40443929</pub-id>
          <pub-id pub-id-type="pmcid">PMC12119545</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref65">
        <label>65</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Alabdulmohsen</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Almahmudi</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Alhashim</surname>
              <given-names>JN</given-names>
            </name>
            <name name-style="western">
              <surname>Almahdi</surname>
              <given-names>MH</given-names>
            </name>
            <name name-style="western">
              <surname>Alkishy</surname>
              <given-names>EF</given-names>
            </name>
            <name name-style="western">
              <surname>Almossabeh</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Alkhalifah</surname>
              <given-names>SA</given-names>
            </name>
          </person-group>
          <article-title>Is ChatGPT a reliable source of patient information an asthma?</article-title>
          <source>Cureus</source>
          <year>2024</year>
          <volume>16</volume>
          <issue>7</issue>
          <fpage>e64114</fpage>
          <pub-id pub-id-type="doi">10.7759/cureus.64114</pub-id>
          <pub-id pub-id-type="medline">39119408</pub-id>
          <pub-id pub-id-type="pmcid">PMC11306643</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref66">
        <label>66</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zalzal</surname>
              <given-names>HG</given-names>
            </name>
            <name name-style="western">
              <surname>Abraham</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>RK</given-names>
            </name>
          </person-group>
          <article-title>Can ChatGPT help patients answer their otolaryngology questions?</article-title>
          <source>Laryngoscope Investig Otolaryngol</source>
          <year>2024</year>
          <volume>9</volume>
          <issue>1</issue>
          <fpage>e1193</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38362184"/>
          </comment>
          <pub-id pub-id-type="doi">10.1002/lio2.1193</pub-id>
          <pub-id pub-id-type="medline">38362184</pub-id>
          <pub-id pub-id-type="pii">LIO21193</pub-id>
          <pub-id pub-id-type="pmcid">PMC10866598</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref67">
        <label>67</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Liau</surname>
              <given-names>ZQG</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>KLM</given-names>
            </name>
            <name name-style="western">
              <surname>Chua</surname>
              <given-names>WL</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the accuracy and relevance of ChatGPT responses to frequently asked questions regarding total knee replacement</article-title>
          <source>Knee Surg Relat Res</source>
          <year>2024</year>
          <volume>36</volume>
          <issue>1</issue>
          <fpage>15</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://kneesurgrelatres.biomedcentral.com/articles/10.1186/s43019-024-00218-5"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s43019-024-00218-5</pub-id>
          <pub-id pub-id-type="medline">38566254</pub-id>
          <pub-id pub-id-type="pii">10.1186/s43019-024-00218-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC10986046</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref68">
        <label>68</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>King</surname>
              <given-names>RC</given-names>
            </name>
            <name name-style="western">
              <surname>Samaan</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>Yeo</surname>
              <given-names>YH</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Kunkel</surname>
              <given-names>DC</given-names>
            </name>
            <name name-style="western">
              <surname>Habib</surname>
              <given-names>AA</given-names>
            </name>
            <name name-style="western">
              <surname>Ghashghaei</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>A multidisciplinary assessment of chatGPT's knowledge of amyloidosis: observational study</article-title>
          <source>JMIR Cardio</source>
          <year>2024</year>
          <volume>8</volume>
          <fpage>e53421</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://cardio.jmir.org/2024//e53421/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/53421</pub-id>
          <pub-id pub-id-type="medline">38640472</pub-id>
          <pub-id pub-id-type="pii">v8i1e53421</pub-id>
          <pub-id pub-id-type="pmcid">PMC11069089</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref69">
        <label>69</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Dong</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Hong</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Qiu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Kang</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Zhai</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the performance of the language model ChatGPT in responding to common questions of people with epilepsy</article-title>
          <source>Epilepsy Behav</source>
          <year>2024</year>
          <volume>151</volume>
          <fpage>109645</fpage>
          <pub-id pub-id-type="doi">10.1016/j.yebeh.2024.109645</pub-id>
          <pub-id pub-id-type="medline">38244419</pub-id>
          <pub-id pub-id-type="pii">S1525-5050(24)00026-X</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref70">
        <label>70</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Valentini</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Szkandera</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Smolle</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Scheipl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Leithner</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Andreou</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence large language model ChatGPT: is it a trustworthy and reliable source of information for sarcoma patients?</article-title>
          <source>Front Public Health</source>
          <year>2024</year>
          <volume>12</volume>
          <fpage>1303319</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38584922"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fpubh.2024.1303319</pub-id>
          <pub-id pub-id-type="medline">38584922</pub-id>
          <pub-id pub-id-type="pmcid">PMC10995284</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref71">
        <label>71</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ayık</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Ercan</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Demirtaş</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yıldırım</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Çakmak</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of ChatGPT-4o's answers to questions about hip arthroscopy from the patient perspective</article-title>
          <source>Jt Dis Relat Surg</source>
          <year>2025</year>
          <volume>36</volume>
          <issue>1</issue>
          <fpage>193</fpage>
          <lpage>199</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://jointdrs.org/full-text/1664"/>
          </comment>
          <pub-id pub-id-type="doi">10.52312/jdrs.2025.1961</pub-id>
          <pub-id pub-id-type="medline">39719917</pub-id>
          <pub-id pub-id-type="pii">jdrs.2025.1961</pub-id>
          <pub-id pub-id-type="pmcid">PMC11734852</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref72">
        <label>72</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kuşcu</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Pamuk</surname>
              <given-names>AE</given-names>
            </name>
            <name name-style="western">
              <surname>Sütay Süslü</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Hosal</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Is chatGPT accurate and reliable in answering questions regarding head and neck cancer?</article-title>
          <source>Front Oncol</source>
          <year>2023</year>
          <volume>13</volume>
          <fpage>1256459</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38107064"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fonc.2023.1256459</pub-id>
          <pub-id pub-id-type="medline">38107064</pub-id>
          <pub-id pub-id-type="pmcid">PMC10722294</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref73">
        <label>73</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lo Bianco</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Cascella</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Day</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kapural</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Robinson</surname>
              <given-names>CL</given-names>
            </name>
            <name name-style="western">
              <surname>Sinagra</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Reliability, accuracy, and comprehensibility of AI-based responses to common patient questions regarding spinal cord stimulation</article-title>
          <source>J Clin Med</source>
          <year>2025</year>
          <volume>14</volume>
          <issue>5</issue>
          <fpage>1453</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=jcm14051453"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/jcm14051453</pub-id>
          <pub-id pub-id-type="medline">40094896</pub-id>
          <pub-id pub-id-type="pii">jcm14051453</pub-id>
          <pub-id pub-id-type="pmcid">PMC11899866</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref74">
        <label>74</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Aras</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Çalışkan</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Assessing the reliability and usefulness of ChatGPT responses on intermittent catheterization queries: a critical analysis</article-title>
          <source>Int J of Uro Nursing</source>
          <year>2024</year>
          <volume>18</volume>
          <issue>3</issue>
          <fpage>e12428</fpage>
          <pub-id pub-id-type="doi">10.1111/ijun.12428</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref75">
        <label>75</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Magruder</surname>
              <given-names>ML</given-names>
            </name>
            <name name-style="western">
              <surname>Rodriguez</surname>
              <given-names>AN</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>JCJ</given-names>
            </name>
            <name name-style="western">
              <surname>Erez</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Piuzzi</surname>
              <given-names>NS</given-names>
            </name>
            <name name-style="western">
              <surname>Scuderi</surname>
              <given-names>GR</given-names>
            </name>
            <name name-style="western">
              <surname>Slover</surname>
              <given-names>JD</given-names>
            </name>
            <name name-style="western">
              <surname>Oh</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Schwarzkopf</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>AF</given-names>
            </name>
            <name name-style="western">
              <surname>Iorio</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Goodman</surname>
              <given-names>SB</given-names>
            </name>
            <name name-style="western">
              <surname>Mont</surname>
              <given-names>MA</given-names>
            </name>
          </person-group>
          <article-title>Assessing ability for ChatGPT to answer total knee arthroplasty-related questions</article-title>
          <source>J Arthroplasty</source>
          <year>2024</year>
          <volume>39</volume>
          <issue>8</issue>
          <fpage>2022</fpage>
          <lpage>2027</lpage>
          <pub-id pub-id-type="doi">10.1016/j.arth.2024.02.023</pub-id>
          <pub-id pub-id-type="medline">38364879</pub-id>
          <pub-id pub-id-type="pii">S0883-5403(24)00122-0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref76">
        <label>76</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chatzopoulos</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Koidou</surname>
              <given-names>VP</given-names>
            </name>
            <name name-style="western">
              <surname>Tsalikis</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Kaklamanos</surname>
              <given-names>EG</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of large language model performance in answering clinical questions on periodontal furcation defect management</article-title>
          <source>Dent J (Basel)</source>
          <year>2025</year>
          <volume>13</volume>
          <issue>6</issue>
          <fpage>271</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=dj13060271"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/dj13060271</pub-id>
          <pub-id pub-id-type="medline">40559174</pub-id>
          <pub-id pub-id-type="pii">dj13060271</pub-id>
          <pub-id pub-id-type="pmcid">PMC12191798</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref77">
        <label>77</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Abdelatif</surname>
              <given-names>NMN</given-names>
            </name>
            <name name-style="western">
              <surname>Arunakul</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Borbon</surname>
              <given-names>CAV</given-names>
            </name>
            <name name-style="western">
              <surname>Chong</surname>
              <given-names>KW</given-names>
            </name>
            <name name-style="western">
              <surname>Chow</surname>
              <given-names>MW</given-names>
            </name>
            <name name-style="western">
              <surname>Hua</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Oji</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Ahumada</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Siu</surname>
              <given-names>KM</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>KJ</given-names>
            </name>
            <name name-style="western">
              <surname>Tanaka</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Taniguchi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Yung</surname>
              <given-names>PS</given-names>
            </name>
            <name name-style="western">
              <surname>Ling</surname>
              <given-names>SK</given-names>
            </name>
          </person-group>
          <article-title>Are large language model-based chatbots effective in providing reliable medical advice for achilles tendinopathy? An international multispecialist evaluation</article-title>
          <source>Orthop J Sports Med</source>
          <year>2025</year>
          <volume>13</volume>
          <issue>4</issue>
          <fpage>23259671251332596</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/23259671251332596?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/23259671251332596</pub-id>
          <pub-id pub-id-type="medline">40322749</pub-id>
          <pub-id pub-id-type="pii">10.1177_23259671251332596</pub-id>
          <pub-id pub-id-type="pmcid">PMC12046157</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref78">
        <label>78</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mykhalko</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Dyditska</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Balatska</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Filak</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Rubtsova</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>AI-driven rehabilitation: evaluation of ChatGPT-4o for generating personalized physical rehabilitation plans in comorbid patients</article-title>
          <source>Wiad Lek</source>
          <year>2025</year>
          <volume>78</volume>
          <issue>4</issue>
          <fpage>753</fpage>
          <lpage>759</lpage>
          <pub-id pub-id-type="doi">10.36740/WLek/203850</pub-id>
          <pub-id pub-id-type="medline">40367458</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref79">
        <label>79</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hoang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Liou</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Rosenberg</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Zaidat</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Duey</surname>
              <given-names>AH</given-names>
            </name>
            <name name-style="western">
              <surname>Shrestha</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Ahmed</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>Cho</surname>
              <given-names>SK</given-names>
            </name>
          </person-group>
          <article-title>An analysis of ChatGPT recommendations for the diagnosis and treatment of cervical radiculopathy</article-title>
          <source>J Neurosurg Spine</source>
          <year>2024</year>
          <volume>41</volume>
          <issue>3</issue>
          <fpage>385</fpage>
          <lpage>395</lpage>
          <pub-id pub-id-type="doi">10.3171/2024.4.SPINE231148</pub-id>
          <pub-id pub-id-type="medline">38941643</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref80">
        <label>80</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Naldi</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Bettoli</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Santoro</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Valetto</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Bolzon</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cassalia</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Cazzaniga</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cima</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Danese</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Emendi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ponzano</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Scarpa</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Dri</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Application of chatGPT as a content generation tool in continuing medical education: acne as a test topic</article-title>
          <source>Dermatol Reports</source>
          <year>2025</year>
          <volume>17</volume>
          <issue>2</issue>
          <fpage>10138</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://boris-portal.unibe.ch/handle/20.500.12422/205833"/>
          </comment>
          <pub-id pub-id-type="doi">10.4081/dr.2024.10138</pub-id>
          <pub-id pub-id-type="medline">39969058</pub-id>
          <pub-id pub-id-type="pmcid">PMC12210357</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref81">
        <label>81</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hack</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Alsleibi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Saleh</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Alon</surname>
              <given-names>EE</given-names>
            </name>
            <name name-style="western">
              <surname>Rabinovics</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Remer</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Are chatbots a reliable source for patient frequently asked questions on neck masses?</article-title>
          <source>Eur Arch Otorhinolaryngol</source>
          <year>2025</year>
          <volume>282</volume>
          <issue>8</issue>
          <fpage>4273</fpage>
          <lpage>4282</lpage>
          <pub-id pub-id-type="doi">10.1007/s00405-025-09433-6</pub-id>
          <pub-id pub-id-type="medline">40307608</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00405-025-09433-6</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref82">
        <label>82</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>David</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zloto</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Katz</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Huna-Baron</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Vishnevskia-Dai</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Armarnik</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zauberman</surname>
              <given-names>NA</given-names>
            </name>
            <name name-style="western">
              <surname>Barnir</surname>
              <given-names>EM</given-names>
            </name>
            <name name-style="western">
              <surname>Singer</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Hostovsky</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Klang</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>The use of artificial intelligence based chat bots in ophthalmology triage</article-title>
          <source>Eye (Lond)</source>
          <year>2025</year>
          <volume>39</volume>
          <issue>4</issue>
          <fpage>785</fpage>
          <lpage>789</lpage>
          <pub-id pub-id-type="doi">10.1038/s41433-024-03488-1</pub-id>
          <pub-id pub-id-type="medline">39592814</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41433-024-03488-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC11885819</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref83">
        <label>83</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Patel</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Ajumobi</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the reliability of OpenAI's ChatGPT-4 in providing pre-colonoscopy patient guidance</article-title>
          <source>Cureus</source>
          <year>2025</year>
          <volume>17</volume>
          <issue>6</issue>
          <fpage>e86512</fpage>
          <pub-id pub-id-type="doi">10.7759/cureus.86512</pub-id>
          <pub-id pub-id-type="medline">40698206</pub-id>
          <pub-id pub-id-type="pmcid">PMC12280836</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref84">
        <label>84</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gomez-Cabello</surname>
              <given-names>CA</given-names>
            </name>
            <name name-style="western">
              <surname>Borna</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pressman</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Haider</surname>
              <given-names>SA</given-names>
            </name>
            <name name-style="western">
              <surname>Forte</surname>
              <given-names>AJ</given-names>
            </name>
          </person-group>
          <article-title>Large language models for intraoperative decision support in plastic surgery: a comparison between ChatGPT-4 and Gemini</article-title>
          <source>Medicina (Kaunas)</source>
          <year>2024</year>
          <volume>60</volume>
          <issue>6</issue>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=medicina60060957"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/medicina60060957</pub-id>
          <pub-id pub-id-type="medline">38929573</pub-id>
          <pub-id pub-id-type="pii">medicina60060957</pub-id>
          <pub-id pub-id-type="pmcid">PMC11205293</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref85">
        <label>85</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ah-Yan</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Boissonnault</surname>
              <given-names>È</given-names>
            </name>
            <name name-style="western">
              <surname>Boudier-Revéret</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Mares</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Impact of artificial intelligence in managing musculoskeletal pathologies in physiatry: a qualitative observational study evaluating the potential use of ChatGPT versus Copilot for patient information and clinical advice on low back pain</article-title>
          <source>J Yeungnam Med Sci</source>
          <year>2025</year>
          <volume>42</volume>
          <fpage>11</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.e-jyms.org/journal/view.php?doi=10.12701/jyms.2024.01151"/>
          </comment>
          <pub-id pub-id-type="doi">10.12701/jyms.2024.01151</pub-id>
          <pub-id pub-id-type="medline">39610054</pub-id>
          <pub-id pub-id-type="pii">jyms.2024.01151</pub-id>
          <pub-id pub-id-type="pmcid">PMC11812099</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref86">
        <label>86</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gumilar</surname>
              <given-names>KE</given-names>
            </name>
            <name name-style="western">
              <surname>Indraprasta</surname>
              <given-names>BR</given-names>
            </name>
            <name name-style="western">
              <surname>Faridzi</surname>
              <given-names>AS</given-names>
            </name>
            <name name-style="western">
              <surname>Wibowo</surname>
              <given-names>BM</given-names>
            </name>
            <name name-style="western">
              <surname>Herlambang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Rahestyningtyas</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Irawan</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Tambunan</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Bustomi</surname>
              <given-names>AF</given-names>
            </name>
            <name name-style="western">
              <surname>Brahmantara</surname>
              <given-names>BN</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Hsu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Pramuditya</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Putra</surname>
              <given-names>VGE</given-names>
            </name>
            <name name-style="western">
              <surname>Nugroho</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Mulawardhana</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Tjokroprawiro</surname>
              <given-names>BA</given-names>
            </name>
            <name name-style="western">
              <surname>Hedianto</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Ibrahim</surname>
              <given-names>IH</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liao</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Assessment of large language models (LLMs) in decision-making support for gynecologic oncology</article-title>
          <source>Comput Struct Biotechnol J</source>
          <year>2024</year>
          <volume>23</volume>
          <fpage>4019</fpage>
          <lpage>4026</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2001-0370(24)00370-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.csbj.2024.10.050</pub-id>
          <pub-id pub-id-type="medline">39610903</pub-id>
          <pub-id pub-id-type="pii">S2001-0370(24)00370-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC11603009</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref87">
        <label>87</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Vitale</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Fekete</surname>
              <given-names>TF</given-names>
            </name>
            <name name-style="western">
              <surname>Haschtmann</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Reitmeir</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ropelato</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Puhakka</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Galbusera</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Loibl</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Are large language models valid tools for patient information on lumbar disc herniation? The spine surgeons' perspective</article-title>
          <source>Brain Spine</source>
          <year>2024</year>
          <volume>4</volume>
          <fpage>102804</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2772-5294(24)00060-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.bas.2024.102804</pub-id>
          <pub-id pub-id-type="medline">38706800</pub-id>
          <pub-id pub-id-type="pii">S2772-5294(24)00060-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC11067000</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref88">
        <label>88</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Vitale</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Galbusera</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Fekete</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Boissiere</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Charles</surname>
              <given-names>YP</given-names>
            </name>
            <name name-style="western">
              <surname>Yucekul</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Yilgor</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Núñez-Pereira</surname>
              <given-names>Susana</given-names>
            </name>
            <name name-style="western">
              <surname>Haddad</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Gomez-Rice</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mehta</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Pizones</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Pellisé</surname>
              <given-names>Ferran</given-names>
            </name>
            <name name-style="western">
              <surname>Obeid</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Alanay</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kleinstück</surname>
              <given-names>Frank</given-names>
            </name>
            <name name-style="western">
              <surname>Loibl</surname>
              <given-names>M</given-names>
            </name>
            <collab>ESSG European Spine Study Group</collab>
          </person-group>
          <article-title>Is the information provided by large language models valid in educating patients about adolescent idiopathic scoliosis? An evaluation of content, clarity, and empathy : the perspective of the European Spine Study Group</article-title>
          <source>Spine Deform</source>
          <year>2025</year>
          <month>03</month>
          <volume>13</volume>
          <issue>2</issue>
          <fpage>361</fpage>
          <lpage>372</lpage>
          <pub-id pub-id-type="doi">10.1007/s43390-024-00955-3</pub-id>
          <pub-id pub-id-type="medline">39495402</pub-id>
          <pub-id pub-id-type="pii">10.1007/s43390-024-00955-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC11893626</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref89">
        <label>89</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sivaramakrishnan</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Almuqahwi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ansari</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lubbad</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Alagamawy</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Sridharan</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Assessing the power of AI: a comparative evaluation of large language models in generating patient education materials in dentistry</article-title>
          <source>BDJ Open</source>
          <year>2025</year>
          <month>06</month>
          <day>18</day>
          <volume>11</volume>
          <issue>1</issue>
          <fpage>59</fpage>
          <pub-id pub-id-type="doi">10.1038/s41405-025-00349-1</pub-id>
          <pub-id pub-id-type="medline">40533491</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41405-025-00349-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC12177049</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref90">
        <label>90</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vassis</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Powell</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Petersen</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Barkmann</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Noeldeke</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Kristensen</surname>
              <given-names>KD</given-names>
            </name>
            <name name-style="western">
              <surname>Stoustrup</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Large-language models in orthodontics: assessing reliability and validity of ChatGPT in pretreatment patient education</article-title>
          <source>Cureus</source>
          <year>2024</year>
          <volume>16</volume>
          <issue>8</issue>
          <fpage>e68085</fpage>
          <pub-id pub-id-type="doi">10.7759/cureus.68085</pub-id>
          <pub-id pub-id-type="medline">39347180</pub-id>
          <pub-id pub-id-type="pmcid">PMC11437517</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref91">
        <label>91</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Demir</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of the reliability and readability of answers given by chatbots to frequently asked questions about endophthalmitis: a cross-sectional study on chatbots</article-title>
          <source>Health Informatics J</source>
          <year>2024</year>
          <volume>30</volume>
          <issue>4</issue>
          <fpage>14604582241304679</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/14604582241304679?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/14604582241304679</pub-id>
          <pub-id pub-id-type="medline">39612357</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref92">
        <label>92</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wright</surname>
              <given-names>BM</given-names>
            </name>
            <name name-style="western">
              <surname>Bodnar</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Moore</surname>
              <given-names>AD</given-names>
            </name>
            <name name-style="western">
              <surname>Maseda</surname>
              <given-names>MC</given-names>
            </name>
            <name name-style="western">
              <surname>Kucharik</surname>
              <given-names>MP</given-names>
            </name>
            <name name-style="western">
              <surname>Diaz</surname>
              <given-names>CC</given-names>
            </name>
            <name name-style="western">
              <surname>Schmidt</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Mir</surname>
              <given-names>HR</given-names>
            </name>
          </person-group>
          <article-title>Is ChatGPT a trusted source of information for total hip and knee arthroplasty patients?</article-title>
          <source>Bone Jt Open</source>
          <year>2024</year>
          <volume>5</volume>
          <issue>2</issue>
          <fpage>139</fpage>
          <lpage>146</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38354748"/>
          </comment>
          <pub-id pub-id-type="doi">10.1302/2633-1462.52.BJO-2023-0113.R1</pub-id>
          <pub-id pub-id-type="medline">38354748</pub-id>
          <pub-id pub-id-type="pii">BJO-2023-0113.R1</pub-id>
          <pub-id pub-id-type="pmcid">PMC10867788</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref93">
        <label>93</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Peled</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Sela</surname>
              <given-names>HY</given-names>
            </name>
            <name name-style="western">
              <surname>Weiss</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Grisaru-Granovsky</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Agrawal</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Rottenstreich</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the validity of ChatGPT responses on common obstetric issues: potential clinical applications and implications</article-title>
          <source>Int J Gynaecol Obstet</source>
          <year>2024</year>
          <volume>166</volume>
          <issue>3</issue>
          <fpage>1127</fpage>
          <lpage>1133</lpage>
          <pub-id pub-id-type="doi">10.1002/ijgo.15501</pub-id>
          <pub-id pub-id-type="medline">38523565</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref94">
        <label>94</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kerkütlüoğlu</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kaya</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Gökmen</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Trustworthiness, value, danger, and readability of ChatGPT-generated responses to health questions related to pulmonary arterial hypertension</article-title>
          <source>Cureus</source>
          <year>2024</year>
          <volume>16</volume>
          <issue>10</issue>
          <fpage>e71472</fpage>
          <pub-id pub-id-type="doi">10.7759/cureus.71472</pub-id>
          <pub-id pub-id-type="medline">39544545</pub-id>
          <pub-id pub-id-type="pmcid">PMC11560386</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref95">
        <label>95</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Salmi</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lewis</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Clarke</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Dong</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Fischmann</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>McIntosh</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Sarabu</surname>
              <given-names>CR</given-names>
            </name>
            <name name-style="western">
              <surname>DesRoches</surname>
              <given-names>CM</given-names>
            </name>
          </person-group>
          <article-title>A proof-of-concept study for patient use of open notes with large language models</article-title>
          <source>JAMIA Open</source>
          <year>2025</year>
          <volume>8</volume>
          <issue>2</issue>
          <fpage>ooaf021</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/jamiaopen/article-lookup/doi/10.1093/jamiaopen/ooaf021"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf021</pub-id>
          <pub-id pub-id-type="medline">40206786</pub-id>
          <pub-id pub-id-type="pii">ooaf021</pub-id>
          <pub-id pub-id-type="pmcid">PMC11980777</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref96">
        <label>96</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Temizsoy Korkmaz</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Ok</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Karip</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Keleş</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>A structured evaluation of LLM-generated step-by-step instructions in cadaveric brachial plexus dissection</article-title>
          <source>BMC Med Educ</source>
          <year>2025</year>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>903</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmededuc.biomedcentral.com/articles/10.1186/s12909-025-07493-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12909-025-07493-0</pub-id>
          <pub-id pub-id-type="medline">40598351</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12909-025-07493-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC12211967</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref97">
        <label>97</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Shi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>The role of ChatGPT-4o in differential diagnosis and management of vertigo-related disorders</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <volume>15</volume>
          <issue>1</issue>
          <fpage>18688</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-025-96309-8"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-025-96309-8</pub-id>
          <pub-id pub-id-type="medline">40437044</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-025-96309-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC12119837</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref98">
        <label>98</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Birsel</surname>
              <given-names>SE</given-names>
            </name>
            <name name-style="western">
              <surname>Oto</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Görgün</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>İnan</surname>
              <given-names>İH</given-names>
            </name>
            <name name-style="western">
              <surname>Şeker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>İnan</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Is it a pediatric orthopaedic urgency or not? Can chatGPT answer this question?</article-title>
          <source>J Orthop Surg Res</source>
          <year>2025</year>
          <volume>20</volume>
          <issue>1</issue>
          <fpage>567</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://josr-online.biomedcentral.com/articles/10.1186/s13018-025-05981-z"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s13018-025-05981-z</pub-id>
          <pub-id pub-id-type="medline">40462187</pub-id>
          <pub-id pub-id-type="pii">10.1186/s13018-025-05981-z</pub-id>
          <pub-id pub-id-type="pmcid">PMC12135399</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref99">
        <label>99</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>George</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Using Google web search to analyze and evaluate the application of ChatGPT in femoroacetabular impingement syndrome</article-title>
          <source>Front Public Health</source>
          <year>2024</year>
          <volume>12</volume>
          <fpage>1412063</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38883198"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fpubh.2024.1412063</pub-id>
          <pub-id pub-id-type="medline">38883198</pub-id>
          <pub-id pub-id-type="pmcid">PMC11176516</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref100">
        <label>100</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Calabrese</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Maselli</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Maida</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Barbaro</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Morais</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Nardone</surname>
              <given-names>OM</given-names>
            </name>
            <name name-style="western">
              <surname>Sinagra</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Di Mitri</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Sferrazza</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Unveiling the effectiveness of Chat-GPT 4.0, an artificial intelligence conversational tool, for addressing common patient queries in gastrointestinal endoscopy</article-title>
          <source>IGIE</source>
          <year>2025</year>
          <volume>4</volume>
          <issue>1</issue>
          <fpage>21</fpage>
          <lpage>25</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2949-7086(25)00012-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.igie.2025.01.012</pub-id>
          <pub-id pub-id-type="medline">41648814</pub-id>
          <pub-id pub-id-type="pii">S2949-7086(25)00012-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC12850834</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref101">
        <label>101</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tam</surname>
              <given-names>TYC</given-names>
            </name>
            <name name-style="western">
              <surname>Sivarajkumar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kapoor</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Stolyar</surname>
              <given-names>AV</given-names>
            </name>
            <name name-style="western">
              <surname>Polanska</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>McCarthy</surname>
              <given-names>KR</given-names>
            </name>
            <name name-style="western">
              <surname>Osterhoudt</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Visweswaran</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Fu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mathur</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Cacciamani</surname>
              <given-names>GE</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title>
          <source>NPJ Digit Med</source>
          <year>2024</year>
          <month>09</month>
          <day>28</day>
          <volume>7</volume>
          <issue>1</issue>
          <fpage>258</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id>
          <pub-id pub-id-type="medline">39333376</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-024-01258-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11437138</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref102">
        <label>102</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hager</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Jungmann</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Holland</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Bhagat</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Hubrecht</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Knauer</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Vielhauer</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Makowski</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Braren</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Kaissis</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Rueckert</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Evaluation and mitigation of the limitations of large language models in clinical decision-making</article-title>
          <source>Nat Med</source>
          <year>2024</year>
          <volume>30</volume>
          <issue>9</issue>
          <fpage>2613</fpage>
          <lpage>2622</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-03097-1</pub-id>
          <pub-id pub-id-type="medline">38965432</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-03097-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC11405275</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref103">
        <label>103</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Schuler</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Jung</surname>
              <given-names>IC</given-names>
            </name>
            <name name-style="western">
              <surname>Zerlik</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hahn</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Sedlmayr</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sedlmayr</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Context factors in clinical decision-making: a scoping review</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2025</year>
          <volume>25</volume>
          <issue>1</issue>
          <fpage>133</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-025-02965-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-025-02965-1</pub-id>
          <pub-id pub-id-type="medline">40098142</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-025-02965-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC11912758</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref104">
        <label>104</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ratwani</surname>
              <given-names>RM</given-names>
            </name>
            <name name-style="western">
              <surname>Bates</surname>
              <given-names>DW</given-names>
            </name>
            <name name-style="western">
              <surname>Classen</surname>
              <given-names>DC</given-names>
            </name>
          </person-group>
          <article-title>Patient safety and artificial intelligence in clinical care</article-title>
          <source>JAMA Health Forum</source>
          <year>2024</year>
          <volume>5</volume>
          <issue>2</issue>
          <fpage>e235514</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://jamanetwork.com/article.aspx?doi=10.1001/jamahealthforum.2023.5514"/>
          </comment>
          <pub-id pub-id-type="doi">10.1001/jamahealthforum.2023.5514</pub-id>
          <pub-id pub-id-type="medline">38393719</pub-id>
          <pub-id pub-id-type="pii">2815239</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref105">
        <label>105</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mohamed</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Kittle</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Nour</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Hamed</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Feeney</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Salsberg</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>An integrative systematic review on interventions to improve layperson's ability to identify trustworthy digital health information</article-title>
          <source>PLOS Digit Health</source>
          <year>2024</year>
          <volume>3</volume>
          <issue>10</issue>
          <fpage>e0000638</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pdig.0000638"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pdig.0000638</pub-id>
          <pub-id pub-id-type="medline">39453891</pub-id>
          <pub-id pub-id-type="pii">PDIG-D-23-00445</pub-id>
          <pub-id pub-id-type="pmcid">PMC11508166</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref106">
        <label>106</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Xiao</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Barandouzi</surname>
              <given-names>ZA</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Brock</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Bold</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of large language models in tailoring educational content for cancer survivors and their caregivers: quality analysis</article-title>
          <source>JMIR Cancer</source>
          <year>2025</year>
          <volume>11</volume>
          <fpage>e67914</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://cancer.jmir.org/2025//e67914/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/67914</pub-id>
          <pub-id pub-id-type="medline">40192716</pub-id>
          <pub-id pub-id-type="pii">v11i1e67914</pub-id>
          <pub-id pub-id-type="pmcid">PMC11995809</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref107">
        <label>107</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Diviani</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>van den Putte</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Giani</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>van Weert</surname>
              <given-names>JC</given-names>
            </name>
          </person-group>
          <article-title>Low health literacy and evaluation of online health information: a systematic review of the literature</article-title>
          <source>J Med Internet Res</source>
          <year>2015</year>
          <volume>17</volume>
          <issue>5</issue>
          <fpage>e112</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2015/5/e112/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/jmir.4018</pub-id>
          <pub-id pub-id-type="medline">25953147</pub-id>
          <pub-id pub-id-type="pii">v17i5e112</pub-id>
          <pub-id pub-id-type="pmcid">PMC4468598</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref108">
        <label>108</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Loomba</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>de Figueiredo</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Piatek</surname>
              <given-names>SJ</given-names>
            </name>
            <name name-style="western">
              <surname>de Graaf</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Larson</surname>
              <given-names>HJ</given-names>
            </name>
          </person-group>
          <article-title>Measuring the impact of COVID-19 vaccine misinformation on vaccination intent in the UK and USA</article-title>
          <source>Nat Hum Behav</source>
          <year>2021</year>
          <volume>5</volume>
          <issue>3</issue>
          <fpage>337</fpage>
          <lpage>348</lpage>
          <pub-id pub-id-type="doi">10.1038/s41562-021-01056-1</pub-id>
          <pub-id pub-id-type="medline">33547453</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41562-021-01056-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref109">
        <label>109</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <collab>Tabassi E</collab>
          </person-group>
          <article-title>Artificial intelligence risk management framework (AI RMF 1.0)</article-title>
          <source>National Institute of Standards and Technology</source>
          <year>2023</year>
          <access-date>2026-08-08</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.6028/NIST.AI.100-1">https://doi.org/10.6028/NIST.AI.100-1</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref110">
        <label>110</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nickel</surname>
              <given-names>PJ</given-names>
            </name>
          </person-group>
          <article-title>Trust in medical artificial intelligence: a discretionary account</article-title>
          <source>Ethics Inf Technol</source>
          <year>2022</year>
          <volume>24</volume>
          <issue>1</issue>
          <fpage>7</fpage>
          <pub-id pub-id-type="doi">10.1007/s10676-022-09630-5</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref111">
        <label>111</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Joshi</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of large language models: review of metrics, applications, and methodologies</article-title>
          <source>Preprints</source>
          <comment>Preprint posted online on April 7, 2025</comment>
          <pub-id pub-id-type="doi">10.20944/preprints202504.0369.v2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref112">
        <label>112</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Shin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Cho</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Analyzing evaluation methods for large language models in the medical field: a scoping review</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2024</year>
          <volume>24</volume>
          <issue>1</issue>
          <fpage>366</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-024-02709-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-024-02709-7</pub-id>
          <pub-id pub-id-type="medline">39614219</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-024-02709-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11606129</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref113">
        <label>113</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Koo</surname>
              <given-names>TK</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>MY</given-names>
            </name>
          </person-group>
          <article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title>
          <source>J Chiropr Med</source>
          <year>2016</year>
          <volume>15</volume>
          <issue>2</issue>
          <fpage>155</fpage>
          <lpage>163</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/27330520"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id>
          <pub-id pub-id-type="medline">27330520</pub-id>
          <pub-id pub-id-type="pii">S1556-3707(16)00015-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC4913118</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref114">
        <label>114</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bouchez</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Cagnon</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hamouche</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Majdoub</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Charlet</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Schuers</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Interprofessional clinical decision-making process in health: a scoping review</article-title>
          <source>J Adv Nurs</source>
          <year>2024</year>
          <volume>80</volume>
          <issue>3</issue>
          <fpage>884</fpage>
          <lpage>907</lpage>
          <pub-id pub-id-type="doi">10.1111/jan.15865</pub-id>
          <pub-id pub-id-type="medline">37705486</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref115">
        <label>115</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kottner</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Audigé</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Brorson</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Donner</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Gajewski</surname>
              <given-names>BJ</given-names>
            </name>
            <name name-style="western">
              <surname>Hróbjartsson</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Roberts</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Shoukri</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Streiner</surname>
              <given-names>DL</given-names>
            </name>
          </person-group>
          <article-title>Guidelines for reporting reliability and agreement studies (GRRAS) were proposed</article-title>
          <source>J Clin Epidemiol</source>
          <year>2011</year>
          <volume>64</volume>
          <issue>1</issue>
          <fpage>96</fpage>
          <lpage>106</lpage>
          <pub-id pub-id-type="doi">10.1016/j.jclinepi.2010.03.002</pub-id>
          <pub-id pub-id-type="medline">21130355</pub-id>
          <pub-id pub-id-type="pii">S0895-4356(10)00097-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref116">
        <label>116</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jacob</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Brasier</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Laurenzi</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Heuss</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mougiakakou</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cöltekin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Peter</surname>
              <given-names>MK</given-names>
            </name>
          </person-group>
          <article-title>AI for IMPACTS framework for evaluating the long-term real-world impacts of AI-powered clinician tools: systematic review and narrative synthesis</article-title>
          <source>J Med Internet Res</source>
          <year>2025</year>
          <volume>27</volume>
          <fpage>e67485</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://boris-portal.unibe.ch/handle/20.500.12422/205123"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/67485</pub-id>
          <pub-id pub-id-type="medline">39909417</pub-id>
          <pub-id pub-id-type="pii">v27i1e67485</pub-id>
          <pub-id pub-id-type="pmcid">PMC11840377</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref117">
        <label>117</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wells</surname>
              <given-names>BJ</given-names>
            </name>
            <name name-style="western">
              <surname>Nguyen</surname>
              <given-names>HM</given-names>
            </name>
            <name name-style="western">
              <surname>McWilliams</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pallini</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Bovi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kuzma</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kramer</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hetherington</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Corn</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Taylor</surname>
              <given-names>YJ</given-names>
            </name>
            <name name-style="western">
              <surname>Cuison</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Gagen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Isreal</surname>
              <given-names>M</given-names>
            </name>
            <collab>FAIR-AI Consortium</collab>
          </person-group>
          <article-title>A practical framework for appropriate implementation and review of artificial intelligence (FAIR-AI) in healthcare</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>08</month>
          <day>11</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>514</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01900-y</pub-id>
          <pub-id pub-id-type="medline">40790350</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01900-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC12340025</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref118">
        <label>118</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Joshi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kale</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chandel</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pal</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Likert scale: explored and explained</article-title>
          <source>Curr J Appl Sci Technol</source>
          <year>2015</year>
          <volume>7</volume>
          <issue>4</issue>
          <fpage>396</fpage>
          <lpage>403</lpage>
          <pub-id pub-id-type="doi">10.9734/bjast/2015/14975</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref119">
        <label>119</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jonsson</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Svingby</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>The use of scoring rubrics: Reliability, validity and educational consequences</article-title>
          <source>Educ Res Rev</source>
          <year>2007</year>
          <month>1</month>
          <volume>2</volume>
          <issue>2</issue>
          <fpage>130</fpage>
          <lpage>144</lpage>
          <pub-id pub-id-type="doi">10.1016/j.edurev.2007.05.002</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref120">
        <label>120</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yeates</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>O'Neill</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Mann</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Eva</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Seeing the same thing differently: mechanisms that contribute to assessor differences in directly-observed performance assessments</article-title>
          <source>Adv Health Sci Educ Theory Pract</source>
          <year>2013</year>
          <volume>18</volume>
          <issue>3</issue>
          <fpage>325</fpage>
          <lpage>341</lpage>
          <pub-id pub-id-type="doi">10.1007/s10459-012-9372-1</pub-id>
          <pub-id pub-id-type="medline">22581567</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref121">
        <label>121</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Berendonk</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Stalmeijer</surname>
              <given-names>RE</given-names>
            </name>
            <name name-style="western">
              <surname>Schuwirth</surname>
              <given-names>LWT</given-names>
            </name>
          </person-group>
          <article-title>Expertise in performance assessment: assessors' perspectives</article-title>
          <source>Adv Health Sci Educ Theory Pract</source>
          <year>2013</year>
          <volume>18</volume>
          <issue>4</issue>
          <fpage>559</fpage>
          <lpage>571</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/22847173"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s10459-012-9392-x</pub-id>
          <pub-id pub-id-type="medline">22847173</pub-id>
          <pub-id pub-id-type="pmcid">PMC3767885</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref122">
        <label>122</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Charnock</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Shepperd</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Needham</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Gann</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>DISCERN: an instrument for judging the quality of written consumer health information on treatment choices</article-title>
          <source>J Epidemiol Community Health</source>
          <year>1999</year>
          <volume>53</volume>
          <issue>2</issue>
          <fpage>105</fpage>
          <lpage>111</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://jech.bmj.com/lookup/pmidlookup?view=long&#38;pmid=10396471"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/jech.53.2.105</pub-id>
          <pub-id pub-id-type="medline">10396471</pub-id>
          <pub-id pub-id-type="pmcid">PMC1756830</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref123">
        <label>123</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shoemaker</surname>
              <given-names>SJ</given-names>
            </name>
            <name name-style="western">
              <surname>Wolf</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Brach</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Development of the patient education materials assessment tool (PEMAT): a new measure of understandability and actionability for print and audiovisual patient information</article-title>
          <source>Patient Educ Couns</source>
          <year>2014</year>
          <volume>96</volume>
          <issue>3</issue>
          <fpage>395</fpage>
          <lpage>403</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/24973195"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.pec.2014.05.027</pub-id>
          <pub-id pub-id-type="medline">24973195</pub-id>
          <pub-id pub-id-type="pii">S0738-3991(14)00233-X</pub-id>
          <pub-id pub-id-type="pmcid">PMC5085258</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref124">
        <label>124</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kogan</surname>
              <given-names>JR</given-names>
            </name>
            <name name-style="western">
              <surname>Conforti</surname>
              <given-names>LN</given-names>
            </name>
            <name name-style="western">
              <surname>Holmboe</surname>
              <given-names>ES</given-names>
            </name>
          </person-group>
          <article-title>Faculty perceptions of frame of reference training to improve workplace-based assessment</article-title>
          <source>J Grad Med Educ</source>
          <year>2023</year>
          <volume>15</volume>
          <issue>1</issue>
          <fpage>81</fpage>
          <lpage>91</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/36817545"/>
          </comment>
          <pub-id pub-id-type="doi">10.4300/JGME-D-22-00287.1</pub-id>
          <pub-id pub-id-type="medline">36817545</pub-id>
          <pub-id pub-id-type="pmcid">PMC9934818</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref125">
        <label>125</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>King</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Phillipi</surname>
              <given-names>CA</given-names>
            </name>
            <name name-style="western">
              <surname>Buchanan</surname>
              <given-names>PM</given-names>
            </name>
            <name name-style="western">
              <surname>Lewin</surname>
              <given-names>LO</given-names>
            </name>
          </person-group>
          <article-title>Self-Directed Rater Training for Pediatric History and Physical Exam Evaluation (P-HAPEE) rubric, a validated written H&#38;P assessment tool</article-title>
          <source>MedEdPORTAL</source>
          <year>2017</year>
          <month>07</month>
          <day>21</day>
          <volume>13</volume>
          <fpage>10603</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/30800805"/>
          </comment>
          <pub-id pub-id-type="doi">10.15766/mep_2374-8265.10603</pub-id>
          <pub-id pub-id-type="medline">30800805</pub-id>
          <pub-id pub-id-type="pmcid">PMC6374741</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref126">
        <label>126</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>van der Lee</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Gatt</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>van Miltenburg</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Krahmer</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Human evaluation of automatically generated text: current trends and best practice guidelines</article-title>
          <source>Computer Speech Lang</source>
          <year>2021</year>
          <month>05</month>
          <volume>67</volume>
          <fpage>101151</fpage>
          <pub-id pub-id-type="doi">10.1016/j.csl.2020.101151</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref127">
        <label>127</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Popović</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Agree to disagree: analysis of inter-annotator disagreements in human evaluation of machine translation output</article-title>
          <year>2021</year>
          <conf-name>Proceedings of the 25th Conference on Computational Natural Language Learning</conf-name>
          <conf-date>November 10-11, 2021</conf-date>
          <conf-loc>Virtual</conf-loc>
          <pub-id pub-id-type="doi">10.18653/v1/2021.conll-1.18</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref128">
        <label>128</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Brunyé</surname>
              <given-names>TT</given-names>
            </name>
          </person-group>
          <article-title>Human evaluation of large language models: a review and protocol selection framework</article-title>
          <source>AI</source>
          <year>2026</year>
          <volume>7</volume>
          <issue>5</issue>
          <fpage>174</fpage>
          <pub-id pub-id-type="doi">10.3390/ai7050174</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref129">
        <label>129</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Rostam</surname>
              <given-names>ZRK</given-names>
            </name>
            <name name-style="western">
              <surname>Takács</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kertész</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Evaluating large language models: a review of metrics and benchmarks</article-title>
          <year>2025</year>
          <conf-name>IEEE 23rd Jubilee International Symposium on Intelligent Systems and Informatics (SISY)</conf-name>
          <conf-date>September 25-27, 2025</conf-date>
          <conf-loc>Subotica, Serbia</conf-loc>
          <publisher-name>IEEE</publisher-name>
          <pub-id pub-id-type="doi">10.1109/SISY67000.2025.11205366</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref130">
        <label>130</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Howcroft</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Belz</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Clinciu</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gkatzia</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Hasan</surname>
              <given-names>SA</given-names>
            </name>
            <name name-style="western">
              <surname>Mahamood</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Twenty years of confusion in human evaluation: NLG needs evaluation sheets and standardised definitions</article-title>
          <year>2020</year>
          <conf-name>Proceedings of the 13th International Conference on Natural Language Generation</conf-name>
          <conf-date>December 15-18, 2020</conf-date>
          <conf-loc>Dublin, Ireland</conf-loc>
          <fpage>169</fpage>
          <lpage>182</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2020.inlg-1.23</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref131">
        <label>131</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Fabbri</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Kryściński</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>McCann</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Xiong</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Socher</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Radev</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Summeval: re-evaluating summarization evaluation</article-title>
          <source>Trans Assoc Computational Linguistics</source>
          <year>2021</year>
          <volume>9</volume>
          <fpage>391</fpage>
          <lpage>409</lpage>
          <pub-id pub-id-type="doi">10.1162/tacl_a_00373</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref132">
        <label>132</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Celikyilmaz</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of text generation: a survey</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on June 6, 2020</comment>
          <pub-id pub-id-type="doi">10.48550/arXiv.2006.14799</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref133">
        <label>133</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Feng</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>WX</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wen</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>A survey on large language model based autonomous agents</article-title>
          <source>Front Comput Sci</source>
          <year>2024</year>
          <month>03</month>
          <day>22</day>
          <volume>18</volume>
          <issue>6</issue>
          <fpage>186345</fpage>
          <pub-id pub-id-type="doi">10.1007/s11704-024-40231-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref134">
        <label>134</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mohammadi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Lo</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yip</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>Evaluation and benchmarking of llm agents: a survey</article-title>
          <year>2025</year>
          <conf-name>31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining</conf-name>
          <conf-date>August 3-7, 2025</conf-date>
          <conf-loc>Toronto, ON</conf-loc>
          <fpage>6129</fpage>
          <lpage>6139</lpage>
          <pub-id pub-id-type="doi">10.1145/3711896.3736570</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref135">
        <label>135</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Ji</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>A survey of LLM-based agents in medicine: how far are we from baymax?</article-title>
          <year>2025</year>
          <conf-name>Findings of the Association for Computational Linguistics: ACL 2025</conf-name>
          <conf-date>July 27 to August 1, 2025</conf-date>
          <conf-loc>Vienna, Austria</conf-loc>
          <fpage>10345</fpage>
          <lpage>10359</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2025.findings-acl.539</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref136">
        <label>136</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vatsal</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Dubey</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Agentic AI in healthcare and medicine: a seven-dimensional taxonomy for empirical evaluation of LLM-based agents</article-title>
          <source>IEEE Access</source>
          <year>2026</year>
          <volume>14</volume>
          <fpage>4840</fpage>
          <lpage>4863</lpage>
          <pub-id pub-id-type="doi">10.1109/access.2026.3651218</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
