<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.0" xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id>
      <journal-title>Journal of Medical Internet Research</journal-title>
      <issn pub-type="epub">1438-8871</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v28i1e96587</article-id>
      <article-id pub-id-type="pmid"/>
      <article-id pub-id-type="doi">10.2196/96587</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Bilingual Performance of Large Language Models in Answering Consumer Health Questions in English and Chinese: Comparative Benchmark Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Steenstra</surname>
            <given-names>Ivan</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Yifeng</surname>
            <given-names>Pan</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Basmaji</surname>
            <given-names>Timotaos</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Wu</surname>
            <given-names>Ziyan</given-names>
          </name>
          <degrees>MD, PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5243-6899</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Xu</surname>
            <given-names>Honglin</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5000-462X</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Feng</surname>
            <given-names>Futai</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-3047-8509</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Zhang</surname>
            <given-names>Shulan</given-names>
          </name>
          <degrees>MD, PhD</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-8469-4737</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Cheng</surname>
            <given-names>Rongrong</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-7933-887X</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Shi</surname>
            <given-names>Tianqi</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0003-0946-4389</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Wang</surname>
            <given-names>Siyu</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0008-3412-8092</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Li</surname>
            <given-names>Yongzhe</given-names>
          </name>
          <degrees>MBBS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Department of Clinical Laboratory, State Key Laboratory of Complex Severe and Rare Diseases</institution>
            <institution>Peking Union Medical College Hospital</institution>
            <institution>Chinese Academy of Medical Sciences and Peking Union Medical College</institution>
            <addr-line>1 Shuaifuyuan Hutong</addr-line>
            <addr-line>Dongcheng District</addr-line>
            <addr-line>Beijing, 100730</addr-line>
            <country>China</country>
            <phone>1 0 69159713</phone>
            <email>yongzhelipumch@126.com</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-8267-0985</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Department of Clinical Laboratory, State Key Laboratory of Complex Severe and Rare Diseases</institution>
        <institution>Peking Union Medical College Hospital</institution>
        <institution>Chinese Academy of Medical Sciences and Peking Union Medical College</institution>
        <addr-line>Beijing</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Department of Rheumatology, Peking Union Medical College Hospital, Peking Union Medical College and Chinese Academy of Medical Sciences</institution>
        <institution>National Clinical Research Center for Dermatologic and Immunologic Diseases (NCRC-DID), Key Laboratory of Rheumatology &amp; Clinical Immunology</institution>
        <institution>Ministry of Education</institution>
        <addr-line>Beijing</addr-line>
        <country>China</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Yongzhe Li <email>yongzhelipumch@126.com</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>29</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>28</volume>
      <elocation-id>e96587</elocation-id>
      <history>
        <date date-type="received">
          <day>30</day>
          <month>3</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>22</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>10</day>
          <month>8</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>10</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Ziyan Wu, Honglin Xu, Futai Feng, Shulan Zhang, Rongrong Cheng, Tianqi Shi, Siyu Wang, Yongzhe Li. Originally published in the Journal of Medical Internet Research (https://www.jmir.org), 29.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on https://www.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.jmir.org/2026/1/e96587" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Large language models (LLMs) are increasingly used as health information intermediaries. Whether they provide comparable accuracy and communication quality across languages has direct implications for health information equity; however, systematic bilingual evaluations remain limited.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to provide a preliminary bilingual benchmark evaluating whether 11 LLMs deliver comparable accuracy and communication quality when answering identical consumer health questions in English and Chinese.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We conducted a controlled evaluation of 11 LLMs (GPT-4.5, Claude Sonnet 4, Gemini 2.5 Flash, Grok 3, DeepSeek R1, Qwen 3, Doubao, Kimi k1.5, Hunyuan T1, ERNIE X1 Turbo, and ChatGLM 4) using 150 binary consumer health questions from the Text Retrieval Conference Health Misinformation Track (2019, 2021, and 2022). All models were accessed through official public-facing web interfaces during May 2025. Models were assessed under 2 full-benchmark prompting conditions (no-context and expert), evaluating accuracy, comprehensiveness, precision, and understandability. Four post hoc error-correction strategies (chain-of-thought [CoT], retrieval-augmented generation [RAG], CoT+RAG, and error attribution) were applied to baseline-incorrect responses. Composite ranking used the technique for order of preference by similarity to ideal solution (TOPSIS), with sensitivity analysis across 3 weighting schemes. Generalized estimating equations and linear mixed models with Benjamini-Hochberg false discovery rate (FDR) correction were applied using a full 3-way interaction specification (model×language×prompt).</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>English and Chinese inputs showed comparable overall accuracy under no-context conditions (1572/1650, 95.27% vs 1548/1650, 93.82%), with no significant language main effect (β=0.00; <italic>P</italic>=.99). No language main effects for any individual model remained significant after FDR correction. TOPSIS analysis identified ChatGPT and Qwen as the most consistently top-ranked models (tier 1 in 12/12 condition×weight−scheme combinations). A model-specific language interaction emerged for communication quality: DeepSeek showed a significant English-language decrement in understandability (β=−0.73; FDR=−0.016), while its decrements in precision and comprehensiveness were not significant after correction. One 3-way interaction survived: Grok showed a disproportionate accuracy reduction when English input and expert prompting were combined (β=−1.88; FDR=−0.022). Among post hoc correction strategies, error attribution achieved the highest correction rate (Δ55.56%), although this condition provided models with privileged information.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>Contemporary LLMs achieved high binary accuracy on consumer health questions in both English and Chinese, with no significant aggregate language effect. The only robust model-specific language interaction was DeepSeek’s English understandability decrement, independently confirmed by TOPSIS tier analysis. These findings suggested that cross-linguistic communication quality concerns were model-specific rather than universal and warrant targeted monitoring.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>large language models</kwd>
        <kwd>consumer health questions</kwd>
        <kwd>bilingual benchmark</kwd>
        <kwd>AI in medicine</kwd>
        <kwd>prompt engineering</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>The rapid integration of large language models (LLMs) into health care communication has positioned these systems as critical intermediaries between medical knowledge and the public. As of 2025, platforms such as ChatGPT, DeepSeek, and Gemini have transitioned from experimental tools to widely accessible resources for health information [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Unlike traditional search engines that retrieve documents for user interpretation, LLMs synthesize information into direct, conversational responses. Comparative evidence indicates that LLMs achieve approximately 80% accuracy in binary health question answering, substantially higher than the 50% to 70% accuracy observed with conventional search engines [<xref ref-type="bibr" rid="ref3">3</xref>]. This performance advantage has accelerated their adoption as sources of medical guidance.</p>
      <p>However, this intermediary role introduces critical risks. Inaccurate medical advice, hallucinated information, or culturally misaligned responses can compromise patient understanding and clinical outcomes [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Most concerning is the potential for systematic variation across languages. When the same health question elicits divergent answers, explanatory depth, or risk framings in different languages, it raises questions about equitable access to reliable health information [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Despite these stakes, existing evaluations have predominantly focused on English-language performance, with limited cross-linguistic comparison of models developed in distinct geopolitical and linguistic contexts [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. This gap is particularly consequential given the emergence of parallel LLM ecosystems optimized for different linguistic environments [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>].</p>
      <p>By 2025, the LLM landscape has evolved into 2 complementary paradigms. US-developed proprietary models, including OpenAI’s ChatGPT, Google’s Gemini, and Anthropic’s Claude, represent high-resource, proprietary systems trained on immense private datasets [<xref ref-type="bibr" rid="ref11">11</xref>]. In parallel, China-developed models, such as DeepSeek, Qwen, and Doubao, have matured into a cost-efficient ecosystem optimized for local linguistic and regulatory contexts, with some models achieving competitive reasoning performance through architectural innovations [<xref ref-type="bibr" rid="ref12">12</xref>]. While English has historically dominated LLM training corpora, increasing multilingual corpus inclusion has enabled models optimized for bilingual communication; however, systematic evaluation of their cross-linguistic reliability remains sparse [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>].</p>
      <p>The clinical implications of language-dependent performance variation are substantial. Inconsistent medical recommendations across languages may erode patient trust, contribute to information asymmetry, and undermine public health communication, particularly in multilingual populations relying on LLMs for consumer and preventive health guidance [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>].</p>
      <p>To address this gap, we conducted a controlled bilingual evaluation of 11 LLMs using 150 consumer health questions presented in both English and Chinese. This study provides a preliminary bilingual benchmark of public-facing LLM responses to medical health misinformation questions. We systematically assessed accuracy, comprehensiveness, precision, and understandability across multiple prompting strategies, including chain-of-thought (CoT) reasoning and retrieval-augmented generation (RAG).</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Study Design and Model Selection</title>
        <p>We conducted a controlled, comparative evaluation of LLMs for bilingual health care question answering during May 1, 2025, to May 31, 2025. Eleven models were evaluated: GPT-4.5 (OpenAI), Gemini 2.5 Flash (Google), Grok 3 (xAI), Claude Sonnet 4 (Anthropic), DeepSeek R1 (DeepSeek), Qwen 3 (Alibaba Group), Doubao (ByteDance), Kimi k1.5 (Moonshot AI), Hunyuan T1 (Tencent), ERNIE X1 Turbo (Baidu), and ChatGLM 4 (Zhipu AI). All models were accessed through official public-facing web interfaces in their most current publicly released versions as of the study date, ensuring the analysis reflected capabilities available to general users. Experimental or restricted-access models were excluded. Each question was asked in a new, independent chat session, ensuring that model responses were not influenced by preceding questions. Each question was queried once per model per condition. For models offering configurable options, conversation memory and personalization features were disabled. For models without user-facing configuration options, default settings were used. Public-facing LLM interfaces do not expose temperature or decoding parameters; therefore, default settings were used for all models, and stochastic variation in outputs cannot be ruled out.</p>
        <p>All models were accessed through their official public-facing interfaces under standard user terms of service. Queries were submitted manually, one at a time, without automated scripting or bulk querying. The evaluation was conducted over the course of 1 month (May 2025), consistent with normal user behavior, in strict compliance with each platform’s use guidelines.</p>
        <p>A detailed table specifying exact model names and versions, platforms, access dates, interface language settings, and configuration options for each model is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Given the rapid iteration cycle of commercial LLMs, results reflect model capabilities at this specific time point and may not generalize to subsequent versions.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>This study was based on the evaluation of LLMs using publicly available databases and did not involve human or animal biological samples or any patient clinical data. Accordingly, an exemption from ethical review was granted by the institutional review board ethics committee of Peking Union Medical College Hospital (I-26ZM0094; <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>).</p>
      </sec>
      <sec>
        <title>Question Set and Bilingual Preparation</title>
        <p>A total of 150 health-related questions were drawn from the Text Retrieval Conference (TREC) Health Misinformation Track (2019, 2021, and 2022; 51 from 2019, 49 from 2021, and 50 from 2022; <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>), excluding COVID-19-only datasets to ensure topic diversity. Each question represented a consumer-facing binary health inquiry with an expert-validated yes or no ground truth. Translation into Chinese was performed after final question selection.</p>
        <p>Questions were prepared in English and Chinese using expert translation followed by independent back-translation to ensure semantic equivalence, yielding a first-pass equivalence rate of 96.67% (580/600). Translation was performed by a bilingual clinical expert (ZW) with medical training in both English and Chinese medical systems. Independent back-translation was performed by a second bilingual expert (HX). Discrepancies were adjudicated by a senior bilingual clinical expert (SZ). All Chinese translations were reviewed for naturalness and cultural or clinical relevance by native Chinese-speaking experts. All Chinese-language questions and prompts were prepared in simplified Chinese. We acknowledge that translated English-origin questions may not reflect health questions naturally asked by Chinese-speaking users, and that semantic equivalence does not guarantee comparable cultural or clinical interpretation. All questions were phrased for lay comprehension while requiring evidence-based reasoning.</p>
        <p>Each model was evaluated under 2 full-benchmark prompting conditions. First, no-context prompt (accuracy 1): question only, for example, “Can cranberries prevent urinary tract infections?” Second, expert prompt (accuracy 2): “Suppose you are a committee of leading scientific experts and medical doctors reviewing the latest and highest quality research. Here is the question: [question]. Choose ‘yes’ or ‘no’ based on your best understanding of current medical practice and literature. If you do not know, you can state that you do not know.” To evaluate error-correction mechanisms, questions answered incorrectly under the no-context condition were subsequently reassessed using four post hoc augmentation strategies: (1) CoT prompt (accuracy 3): instruction to generate step-by-step reasoning before providing final answer; (2) RAG prompt (accuracy 4): top 5 Google Search results retrieved, filtered to 3 most relevant passages, with instruction to base answer primarily on evidence; (3) combined CoT+RAG prompt (accuracy 5): integration of reasoning scaffolds and external evidence; and (4) error attribution prompt (accuracy 6): explicit notification that the model’s prior answer was incorrect, with instruction to reconsider. This condition provided the model with privileged information about the correctness of its prior response and should therefore be interpreted as an artificial diagnostic stress test assessing maximum error-correction capacity, not as a realistic or deployable user-facing prompting strategy. The detailed prompts are presented in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p>
      </sec>
      <sec>
        <title>Evaluation Procedure and Metrics</title>
        <p>The evaluation involved 2 independent reviewers (ZW and HX). Both reviewers independently scored all responses based on predefined metrics. Discrepancies were resolved through structured discussion until consensus was achieved. Responses were presented to evaluators in a randomized order, with model identity concealed until consensus scoring was complete. However, language condition could not be blinded because raters necessarily saw whether responses were in English or Chinese. Additionally, some models may have recognizable stylistic patterns, although the large volume of evaluation (6600 responses) made it difficult for raters to reliably identify individual models. We acknowledge these limits of blinding. Raters evaluated responses in the original language (English responses in English and Chinese responses in Chinese). Both raters were bilingual clinical experts with medical training conducted in Chinese and English-language medical literature proficiency. Interrater reliability was quantified using Cohen κ for categorical ratings (accuracy) and weighted κ for ordinal ratings (comprehensiveness, precision, and understandability), with κ values above 0.75 indicating excellent agreement.</p>
        <p>Model performance was assessed along 4 primary dimensions. Accuracy responses were evaluated for factual correctness relative to authoritative references, including peer-reviewed scientific literature and major health organization publications, using a 2-point scale: correct or incorrect. Responses of “I do not know” or explicit refusals to answer were scored as incorrect. Responses containing both affirmative and negative statements were scored based on the predominant conclusion; when no predominant conclusion could be identified, the response was scored as incorrect. Comprehensiveness measured the scope and depth of each response, scored 1 to 10 with anchors: score of 1 to 3 (response addressed question but provides no explanation or context), score of 4 to 6 (provided some explanation but misses key context or nuance), score of 7 to 9 (provided thorough explanation with appropriate context, examples, and caveats), and score of 10 (comprehensive, well-structured response suitable for lay audiences). Precision assessed the exactness and relevance of the content, evaluating whether responses directly addressed the inquiry without containing hallucinated, tangential, or redundant information, scored 1 to 10 with corresponding anchors. Finally, understandability evaluated the linguistic clarity and logical flow, including use of everyday language, logical organization, and avoidance of unexplained jargon, scored 1 to 10. The full scoring rubric with anchors and examples was provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p>
      </sec>
      <sec>
        <title>RAG and CoT</title>
        <p>To assess whether RAG could mitigate model errors, we applied a 2-stage RAG pipeline to all questions that were initially answered incorrectly by each model under the no-context condition. For each question, we retrieved the top 3 search results using Google Search and filtered these to identify the 3 most relevant evidence passages. Evidence was selected by 1 reviewer based on clinical relevance, source authority (prioritizing peer-reviewed publications, institutional guidelines, and authoritative health organization pages), and directness of evidence. Search terms consisted of the original question text. Representative search queries, retrieved URLs, and selected passages are provided in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>. The final RAG prompt for each model consisted of three components: (1) structured evidence section containing the selected passages, (2) constrained instruction requiring the model to base its answer primarily on the evidence, and (3) a forced-choice yes or no answer. RAG benefit was quantified as Δ-accuracy, defined as the proportion of baseline errors corrected by RAG. In parallel, CoT prompting was applied to all incorrect baseline responses to evaluate whether reasoning augmentation alone was sufficient for error correction. CoT prompts instructed models to display step-by-step reasoning before providing a final answer. However, for commercial models with opaque architectures, the displayed reasoning may represent a post hoc explanation generated to accompany the answer rather than a faithful trace of the model’s internal reasoning process. Our evaluation therefore focused exclusively on whether CoT prompting changed final-answer accuracy, not on the quality or faithfulness of the displayed reasoning steps. Comparative RAG-CoT analysis was used to describe error patterns based on surface-level reasoning outputs. RAG findings should be interpreted as exploratory because the retrieval procedure is inherently nonreproducible: Google search results were dynamic, location-sensitive, and personalization-dependent.</p>
      </sec>
      <sec>
        <title>Statistical Analysis</title>
        <p>Statistical analyses were conducted using Python (version 3.13). Figures were formatted in Adobe Illustrator 2023. Data that did not follow a normal distribution were presented as medians with IQRs. Accuracy comparisons used chi-square and McNemar tests; ordinal metrics were analyzed using Kruskal-Wallis tests. For generalized estimating equations (GEEs): the dependent variable was binary accuracy; independent variables were model (reference: ChatGLM), language (reference: Chinese), and prompt type (reference: no-context); working correlation structure was exchangeable; the clustering unit was question ID; and logit link function was used. For linear mixed models (LMMs): dependent variables were comprehensiveness, precision, and understandability scores (separate models). All interaction terms were prespecified.</p>
        <p>To quantitatively integrate the 4 metrics into a single composite score, the technique for order of preference by similarity to ideal solution (TOPSIS) was used. An equal-weighting scheme was adopted, where a weight coefficient of 0.25 was allocated to accuracy, precision, comprehensiveness, and understandability. Sensitivity analyses were conducted using 2 alternative weighting schemes: communication-weighted (0.25/0.25/0.30/0.20) and safety-weighted (0.50/0.167/0.167/0.167). TOPSIS CIs were estimated using nonparametric bootstrap resampling (1000 iterations) at the question level, with 95% CIs reported. Tier classification used K-means clustering (k=3) on TOPSIS composite scores; the number of clusters was selected based on the elbow method and silhouette analysis. Error-correction strategies were evaluated exclusively on baseline-incorrect responses, thus Δ-accuracy represented within-error-subset correction rate, not an overall accuracy gain. The correction strategies (CoT, RAG, CoT+RAG, and error attribution) were applied exclusively to questions answered incorrectly at baseline. The resulting Δ-accuracy values therefore represented the correction rate within the error subset, not overall accuracy gains, and cannot be directly compared with full-benchmark prompt conditions. Benjamini-Hochberg false discovery rate (FDR) correction was applied to all pairwise and interaction comparisons. Both uncorrected and FDR-corrected <italic>P</italic> values are reported for key findings. Ninety-five percent CIs are reported for key accuracy differences and odds ratios. Statistical significance was defined as <italic>P</italic>&lt;.05.</p>
        <p>The TRIPOD-LLM (Transparent Reporting of a Multivariable Model for Individual Prognosis or Diagnosis–Large Language Model) reporting checklist is presented in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Overall Bilingual Performance and Interrater Reliability</title>
        <p>Eleven LLMs were evaluated on 150 health-related binary questions (drawn from the 2019, 2021, and 2022 TREC datasets) in both English and Chinese, generating 3300 model-question pairs (<xref rid="figure1" ref-type="fig">Figure 1</xref>). Interrater reliability for the evaluation framework was excellent. Cohen κ for accuracy exceeded 0.80 across all conditions (<xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>).</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Study design and evaluation workflow. Schematic diagram created using Figdraw illustrating the 6-step methodological framework used to evaluate large language model (LLM) performance in consumer question-answering tasks. The framework incorporated 2 language conditions (Chinese and English) and 6 distinct prompting strategies. Key methods applied include the technique for order of preference by similarity to ideal solution (TOPSIS) and generalized estimating equations (GEE) for longitudinal analysis. TREC: Text Retrieval Conference.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e96587_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Baseline performance was high overall. Aggregate mean accuracy was 93.82% (1548/1650) for Chinese under no-context conditions and 95.27% (1572/1650) for English under no-context conditions. Under expert prompting, mean accuracy was 91.82% (1515/1650) for Chinese and 91.45% (1509/1650) for English (<xref ref-type="table" rid="table1">Table 1</xref>; <xref rid="figure2" ref-type="fig">Figure 2</xref>). The highest-performing models under the no-context condition were ChatGPT (Chinese: 1584/1650, 95.97%) and Grok (English: 1606/1650, 97.32%), whereas the lowest-performing models were Kimi (Chinese: 1484/1650, 89.94%) and ChatGLM (English: 1463/1650, 88.65%). The high baseline accuracy (&gt;90%) across most models creates ceiling effects that constrain meaningful discrimination of model performance on binary questions.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Model accuracy across years, languages, and prompting conditions.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="140"/>
            <col width="120"/>
            <col width="140"/>
            <col width="150"/>
            <col width="150"/>
            <col width="150"/>
            <col width="150"/>
            <thead>
              <tr valign="top">
                <td>Model</td>
                <td>Language</td>
                <td>Prompt</td>
                <td colspan="3">Year</td>
                <td>Overall, mean (SD)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>2019, mean (SD)</td>
                <td>2021, mean (SD)</td>
                <td>2022, mean (SD)</td>
                <td>
                  <break/>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>ChatGLM</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>90.20 (30.03)</td>
                <td>87.76 (33.12)</td>
                <td>88.00 (32.83)</td>
                <td>88.65 (32.02)</td>
              </tr>
              <tr valign="top">
                <td>ChatGLM</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>83.67 (37.34)</td>
                <td>94.00 (23.99)</td>
                <td>91.90 (26.87)</td>
              </tr>
              <tr valign="top">
                <td>ChatGLM</td>
                <td>English</td>
                <td>Expert</td>
                <td>84.31 (36.73)</td>
                <td>85.71 (35.36)</td>
                <td>96.00 (19.79)</td>
                <td>88.67 (31.58)</td>
              </tr>
              <tr valign="top">
                <td>ChatGLM</td>
                <td>English</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>83.67 (37.34)</td>
                <td>94.00 (23.99)</td>
                <td>91.90 (26.87)</td>
              </tr>
              <tr valign="top">
                <td>ChatGPT</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>94.12 (23.76)</td>
                <td>89.80 (30.58)</td>
                <td>96.00 (19.79)</td>
                <td>93.31 (25.11)</td>
              </tr>
              <tr valign="top">
                <td>ChatGPT</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>93.88 (24.22)</td>
                <td>96.00 (19.79)</td>
                <td>95.97 (19.78)</td>
              </tr>
              <tr valign="top">
                <td>ChatGPT</td>
                <td>English</td>
                <td>Expert</td>
                <td>84.31 (36.73)</td>
                <td>87.76 (33.12)</td>
                <td>100.00 (0.00)</td>
                <td>90.69 (28.55)</td>
              </tr>
              <tr valign="top">
                <td>ChatGPT</td>
                <td>English</td>
                <td>No-context</td>
                <td>94.12 (23.76)</td>
                <td>93.88 (24.22)</td>
                <td>100.00 (0.00)</td>
                <td>96.00 (19.59)</td>
              </tr>
              <tr valign="top">
                <td>Claude</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>88.24 (32.54)</td>
                <td>91.84 (27.66)</td>
                <td>96.00 (19.79)</td>
                <td>92.03 (27.18)</td>
              </tr>
              <tr valign="top">
                <td>Claude</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>100.00 (0.00)</td>
                <td>87.76 (33.12)</td>
                <td>96.00 (19.79)</td>
                <td>94.59 (22.28)</td>
              </tr>
              <tr valign="top">
                <td>Claude</td>
                <td>English</td>
                <td>Expert</td>
                <td>94.12 (23.76)</td>
                <td>89.80 (30.58)</td>
                <td>98.00 (14.14)</td>
                <td>93.97 (23.80)</td>
              </tr>
              <tr valign="top">
                <td>Claude</td>
                <td>English</td>
                <td>No-context</td>
                <td>94.12 (23.76)</td>
                <td>93.88 (24.22)</td>
                <td>100.00 (0.00)</td>
                <td>96.00 (19.59)</td>
              </tr>
              <tr valign="top">
                <td>DeepSeek</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>98.04 (14.00)</td>
                <td>87.76 (33.12)</td>
                <td>96.00 (19.79)</td>
                <td>93.93 (23.70)</td>
              </tr>
              <tr valign="top">
                <td>DeepSeek</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>91.84 (27.66)</td>
                <td>98.00 (14.14)</td>
                <td>95.96 (19.67)</td>
              </tr>
              <tr valign="top">
                <td>DeepSeek</td>
                <td>English</td>
                <td>Expert</td>
                <td>96.08 (19.60)</td>
                <td>85.71 (35.36)</td>
                <td>96.00 (19.79)</td>
                <td>92.60 (25.99)</td>
              </tr>
              <tr valign="top">
                <td>DeepSeek</td>
                <td>English</td>
                <td>No-context</td>
                <td>96.08 (19.60)</td>
                <td>83.67 (37.34)</td>
                <td>100.00 (0.00)</td>
                <td>93.25 (24.35)</td>
              </tr>
              <tr valign="top">
                <td>Doubao</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>92.16 (27.15)</td>
                <td>87.76 (33.12)</td>
                <td>94.00 (23.99)</td>
                <td>91.31 (28.34)</td>
              </tr>
              <tr valign="top">
                <td>Doubao</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>96.08 (19.60)</td>
                <td>93.88 (24.22)</td>
                <td>96.00 (19.79)</td>
                <td>95.32 (21.31)</td>
              </tr>
              <tr valign="top">
                <td>Doubao</td>
                <td>English</td>
                <td>Expert</td>
                <td>92.16 (27.15)</td>
                <td>85.71 (35.36)</td>
                <td>96.00 (19.79)</td>
                <td>91.29 (28.16)</td>
              </tr>
              <tr valign="top">
                <td>Doubao</td>
                <td>English</td>
                <td>No-context</td>
                <td>96.08 (19.60)</td>
                <td>91.84 (27.66)</td>
                <td>98.00 (14.14)</td>
                <td>95.31 (21.21)</td>
              </tr>
              <tr valign="top">
                <td>ERNIE</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>92.16 (27.15)</td>
                <td>73.47 (44.61)</td>
                <td>92.00 (27.40)</td>
                <td>85.88 (34.05)</td>
              </tr>
              <tr valign="top">
                <td>ERNIE</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>85.71 (35.36)</td>
                <td>96.00 (19.79)</td>
                <td>93.25 (24.75)</td>
              </tr>
              <tr valign="top">
                <td>ERNIE</td>
                <td>English</td>
                <td>Expert</td>
                <td>86.27 (34.75)</td>
                <td>83.67 (37.34)</td>
                <td>98.00 (14.14)</td>
                <td>89.31 (30.56)</td>
              </tr>
              <tr valign="top">
                <td>ERNIE</td>
                <td>English</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>89.80 (30.58)</td>
                <td>98.00 (14.14)</td>
                <td>95.28 (21.06)</td>
              </tr>
              <tr valign="top">
                <td>Gemini</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>90.20 (30.03)</td>
                <td>87.76 (33.12)</td>
                <td>96.00 (19.79)</td>
                <td>91.32 (28.23)</td>
              </tr>
              <tr valign="top">
                <td>Gemini</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>96.08 (19.60)</td>
                <td>83.67 (37.34)</td>
                <td>98.00 (14.14)</td>
                <td>92.58 (25.68)</td>
              </tr>
              <tr valign="top">
                <td>Gemini</td>
                <td>English</td>
                <td>Expert</td>
                <td>82.35 (38.50)</td>
                <td>91.84 (27.66)</td>
                <td>94.00 (23.99)</td>
                <td>89.40 (30.67)</td>
              </tr>
              <tr valign="top">
                <td>Gemini</td>
                <td>English</td>
                <td>No-context</td>
                <td>96.08 (19.60)</td>
                <td>93.88 (24.22)</td>
                <td>98.00 (14.14)</td>
                <td>95.99 (19.75)</td>
              </tr>
              <tr valign="top">
                <td>Grok</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>94.12 (23.76)</td>
                <td>93.88 (24.22)</td>
                <td>96.00 (19.79)</td>
                <td>94.67 (22.68)</td>
              </tr>
              <tr valign="top">
                <td>Grok</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>96.08 (19.60)</td>
                <td>85.71 (35.36)</td>
                <td>96.00 (19.79)</td>
                <td>92.60 (25.99)</td>
              </tr>
              <tr valign="top">
                <td>Grok</td>
                <td>English</td>
                <td>Expert</td>
                <td>82.35 (38.50)</td>
                <td>85.71 (35.36)</td>
                <td>98.00 (14.14)</td>
                <td>88.69 (31.27)</td>
              </tr>
              <tr valign="top">
                <td>Grok</td>
                <td>English</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>95.92 (19.99)</td>
                <td>98.00 (14.14)</td>
                <td>97.32 (16.28)</td>
              </tr>
              <tr valign="top">
                <td>Hunyuan</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>96.08 (19.60)</td>
                <td>95.92 (19.99)</td>
                <td>96.00 (19.79)</td>
                <td>96.00 (19.79)</td>
              </tr>
              <tr valign="top">
                <td>Hunyuan</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>89.80 (30.58)</td>
                <td>96.00 (19.79)</td>
                <td>94.61 (22.53)</td>
              </tr>
              <tr valign="top">
                <td>Hunyuan</td>
                <td>English</td>
                <td>Expert</td>
                <td>94.12 (23.76)</td>
                <td>89.80 (30.58)</td>
                <td>96.00 (19.79)</td>
                <td>93.31 (25.11)</td>
              </tr>
              <tr valign="top">
                <td>Hunyuan</td>
                <td>English</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>93.88 (24.22)</td>
                <td>98.00 (14.14)</td>
                <td>96.64 (18.10)</td>
              </tr>
              <tr valign="top">
                <td>Kimi</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>90.20 (30.03)</td>
                <td>85.71 (35.36)</td>
                <td>92.00 (27.40)</td>
                <td>89.30 (31.11)</td>
              </tr>
              <tr valign="top">
                <td>Kimi</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>94.12 (23.76)</td>
                <td>85.71 (35.36)</td>
                <td>90.00 (30.30)</td>
                <td>89.94 (30.18)</td>
              </tr>
              <tr valign="top">
                <td>Kimi</td>
                <td>English</td>
                <td>Expert</td>
                <td>94.12 (23.76)</td>
                <td>89.80 (30.58)</td>
                <td>98.00 (14.14)</td>
                <td>93.97 (23.80)</td>
              </tr>
              <tr valign="top">
                <td>Kimi</td>
                <td>English</td>
                <td>No-context</td>
                <td>96.08 (19.60)</td>
                <td>85.71 (35.36)</td>
                <td>100.00 (0.00)</td>
                <td>93.93 (23.34)</td>
              </tr>
              <tr valign="top">
                <td>Qwen</td>
                <td>Chinese</td>
                <td>Expert</td>
                <td>98.04 (14.00)</td>
                <td>83.67 (37.34)</td>
                <td>98.00 (14.14)</td>
                <td>93.24 (24.43)</td>
              </tr>
              <tr valign="top">
                <td>Qwen</td>
                <td>Chinese</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>87.76 (33.12)</td>
                <td>98.00 (14.14)</td>
                <td>94.60 (22.31)</td>
              </tr>
              <tr valign="top">
                <td>Qwen</td>
                <td>English</td>
                <td>Expert</td>
                <td>98.04 (14.00)</td>
                <td>85.71 (35.36)</td>
                <td>98.00 (14.14)</td>
                <td>93.92 (23.43)</td>
              </tr>
              <tr valign="top">
                <td>Qwen</td>
                <td>English</td>
                <td>No-context</td>
                <td>98.04 (14.00)</td>
                <td>91.84 (27.66)</td>
                <td>98.00 (14.14)</td>
                <td>95.96 (19.67)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Heatmap showing mean accuracy (%) for each model under 4 evaluation conditions. Accuracy values were aggregated across all 3 benchmark datasets (2019, 2021, and 2022) and are displayed on a color gradient from red (low) to green (high).</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e96587_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Statistical analysis identified significant main effects of dataset year (<italic>χ</italic><sup>2</sup><sub>2</sub>=112.3; <italic>P</italic>&lt;.001) and prompting strategy (<italic>χ</italic><sup>2</sup><sub>1</sub>=21.3; <italic>P</italic>&lt;.001) on accuracy. No significant main effect was found for language (<italic>χ</italic><sup>2</sup><sub>1</sub>=0.7; <italic>P</italic>=.41). Differences in accuracy across 2019, 2021, and 2022 question subsets may reflect differences in question difficulty, topic composition, or ground-truth complexity rather than temporal maturation of models (<xref ref-type="supplementary-material" rid="app9">Multimedia Appendices 9</xref> and <xref ref-type="supplementary-material" rid="app10">10</xref>). As all models were evaluated at 1 time point (May 2025), the study cannot infer temporal model maturation from dataset year.</p>
        <p>Under no-context prompting, Chinese-language accuracy varied across dataset-year subsets (<xref ref-type="supplementary-material" rid="app11">Multimedia Appendix 11</xref>). In the 2019 subset, high performance was observed, with Claude achieving 100% (51/51), while Kimi recorded 94.12% (48/51). In the 2021 subset, mean accuracy was lower at 88.13%. Significant gaps emerged, such as the 10.2 percentage point difference between highest-scoring models (ChatGPT and Doubao: 46/49, 93.88%) and lowest-scoring models (ChatGLM and Gemini: 41/49, 83.67%). In the 2022 subset, mean accuracy was higher at 95.82%.</p>
        <p>Under no-context prompting, English-language accuracy showed a similar pattern of variation across dataset-year subsets. In the 2019 subset, high performance was observed, with Qwen, Hunyuan, ERNIE X1 Turbo, ChatGLM and Grok achieving 98.04% (50/51) accuracy, while ChatGPT and Claude dropped to 94.12% (48/51). In the 2021 subset, mean accuracy was lower at 90.72%. The range between highest and lowest performers was 12.25 percentage points. The 2022 subset had the highest accuracy among subsets; 4 models (DeepSeek, Kimi, ChatGPT, and Claude) achieved 100% accuracy.</p>
      </sec>
      <sec>
        <title>GEE and LMM Analysis</title>
        <p>GEE analysis with an exchangeable working correlation structure, clustering on question ID, revealed no significant language main effect on accuracy (β=0.00; <italic>P</italic>=.99; <xref ref-type="table" rid="table2">Table 2</xref>; <xref ref-type="supplementary-material" rid="app12">Multimedia Appendices 12</xref>-<xref ref-type="supplementary-material" rid="app16">16</xref>). No individual model language main effects survived FDR correction (all FDR&gt;0.05). Expert prompt main effect was nonsignificant (β=−0.39; <italic>P</italic>=.17; FDR=−0.47). DeepSeek×English remained significant only for understandability (β=−0.73; FDR=−0.016), not for precision (β=−0.23; FDR=−0.62) or comprehensiveness (β=−0.28; FDR=−0.56). Multiple model×prompt interactions were significant for comprehensiveness and precision (positive β, indicating some models show less degradation under expert prompting). Only one 3-way interaction (model×language×prompt) survived FDR: Grok×English×expert for accuracy (β=−1.88; FDR=0.022).</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Summary of key effects of generalized estimating equation (GEE) and linear mixed model (LMM) analyses.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="260"/>
            <col width="110"/>
            <col width="220"/>
            <col width="110"/>
            <col width="140"/>
            <col width="130"/>
            <thead>
              <tr valign="bottom">
                <td colspan="2">Effect and metric</td>
                <td>Model<sup>a</sup></td>
                <td>β<sup>b</sup> (95% CI)</td>
                <td><italic>P</italic> value (raw)</td>
                <td><italic>P</italic> value (FDR<sup>c</sup>)<sup>d</sup></td>
                <td><italic>P</italic> value (global FDR)<sup>e</sup></td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="7">
                  <bold>Language main effect (English vs Chinese)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Accuracy</td>
                <td>GEE<sup>f</sup></td>
                <td>0.00 (0 to 0)</td>
                <td>—<sup>g</sup></td>
                <td>—</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Comprehensiveness</td>
                <td>LMM<sup>h</sup></td>
                <td>0.01 (−0.32 to 0.35)</td>
                <td>.94</td>
                <td>.97</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td>LMM</td>
                <td>0.01 (−0.33 to 0.35)</td>
                <td>.94</td>
                <td>.97</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Understandability</td>
                <td>LMM</td>
                <td>0.02 (−0.32 to 0.36)</td>
                <td>.91</td>
                <td>.96</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td colspan="7">
                  <bold>Expert prompt (vs no-context)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Accuracy</td>
                <td>GEE</td>
                <td>−0.39 (−0.93 to 0.16)</td>
                <td>.17</td>
                <td>.47</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Comprehensiveness</td>
                <td>LMM</td>
                <td>−0.56 (−0.89 to −0.22)</td>
                <td>.001</td>
                <td>.003</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td>LMM</td>
                <td>−0.66 (−1.00 to −0.32)</td>
                <td>.001</td>
                <td>.001</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Understandability</td>
                <td>LMM</td>
                <td>−0.30 (−0.64 to 0.04)</td>
                <td>.09</td>
                <td>.22</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td colspan="7">
                  <bold>DeepSeek×English (2-way interaction)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Accuracy</td>
                <td>GEE</td>
                <td>−0.54 (−1.29 to 0.21)</td>
                <td>.16</td>
                <td>.47</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Comprehensiveness</td>
                <td>LMM</td>
                <td>−0.28 (−0.75 to 0.19)</td>
                <td>.25</td>
                <td>.46</td>
                <td>.56</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td>LMM</td>
                <td>−0.23 (−0.71 to 0.25)</td>
                <td>.35</td>
                <td>.50</td>
                <td>.62</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Understandability</td>
                <td>LMM</td>
                <td>−0.73 (−1.21 to −0.25)</td>
                <td>.003</td>
                <td>.01</td>
                <td>.02</td>
              </tr>
              <tr valign="top">
                <td colspan="7">
                  <bold>Grok×English×expert (3-way interaction)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Accuracy</td>
                <td>GEE</td>
                <td>−1.88 (−2.94 to −0.82)</td>
                <td>.001</td>
                <td>.02</td>
                <td>—</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Comprehensiveness</td>
                <td>LMM</td>
                <td>−0.76 (−1.43 to −0.09)</td>
                <td>.03</td>
                <td>.07</td>
                <td>.11</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td>LMM</td>
                <td>−0.78 (−1.46 to −0.10)</td>
                <td>.03</td>
                <td>.06</td>
                <td>.11</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Understandability</td>
                <td>LMM</td>
                <td>−0.79 (−1.47 to −0.11)</td>
                <td>.02</td>
                <td>.06</td>
                <td>.10</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>Reference categories: model—ChatGLM, language—Chinese, and prompt—no-context.</p>
            </fn>
            <fn id="table2fn2">
              <p><sup>b</sup>β: regression coefficient on the log-odds scale (GEE) or the raw score scale (LMM).</p>
            </fn>
            <fn id="table2fn3">
              <p><sup>c</sup>FDR: false discovery rate.</p>
            </fn>
            <fn id="table2fn4">
              <p><sup>d</sup>Benjamini-Hochberg false discovery rate–corrected <italic>P</italic> value within each metric.</p>
            </fn>
            <fn id="table2fn5">
              <p><sup>e</sup>Correction was applied jointly across all 4 metrics for the interaction terms.</p>
            </fn>
            <fn id="table2fn6">
              <p><sup>f</sup>GEE: binary accuracy, logit link, and exchangeable correlation clustered on question ID.</p>
            </fn>
            <fn id="table2fn7">
              <p><sup>g</sup>Global false discovery rate was not applicable.</p>
            </fn>
            <fn id="table2fn8">
              <p><sup>h</sup>LMM: comprehensiveness, precision, and understandability as separate dependent variables.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>TOPSIS With Sensitivity Analysis</title>
        <p>Under the equal-weighting scheme for Chinese no-context conditions (<xref ref-type="table" rid="table3">Table 3</xref>; <xref rid="figure3" ref-type="fig">Figures 3</xref> and <xref rid="figure4" ref-type="fig">4</xref>), ChatGPT achieved the highest TOPSIS score (0.95, 95% CI 0.88-1.00), followed by DeepSeek (0.93, 95% CI 0.82-1.00), Qwen (0.90, 95% CI 0.76-0.98), and Doubao (0.88, 95% CI 0.72-1.00). Under English no-context conditions, ChatGPT again ranked the highest (0.97, 95% CI 0.89-1.00).</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Technique for order of preference by similarity to ideal solution (TOPSIS) composite performance scores and tier classifications across conditions.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="330"/>
            <col width="200"/>
            <col width="260"/>
            <col width="110"/>
            <col width="100"/>
            <thead>
              <tr valign="top">
                <td>Condition</td>
                <td>Model</td>
                <td>TOPSIS score (95% CI)</td>
                <td>Tier</td>
                <td>Rank</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>DeepSeek</td>
                <td>0.97 (0.82-1.00)</td>
                <td>Tier 1</td>
                <td>1</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>Qwen</td>
                <td>0.95 (0.84-0.98)</td>
                <td>Tier 1</td>
                <td>2</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>ChatGPT</td>
                <td>0.90 (0.79-0.97)</td>
                <td>Tier 1</td>
                <td>3</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>Grok</td>
                <td>0.75 (0.46-0.80)</td>
                <td>Tier 2</td>
                <td>4</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>Doubao</td>
                <td>0.70 (0.39-0.76)</td>
                <td>Tier 2</td>
                <td>5</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>Hunyuan</td>
                <td>0.70 (0.36-0.74)</td>
                <td>Tier 2</td>
                <td>6</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>Gemini</td>
                <td>0.65 (0.25-0.70)</td>
                <td>Tier 2</td>
                <td>7</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>Claude</td>
                <td>0.54 (0.03-0.58)</td>
                <td>Tier 2</td>
                <td>8</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>Kimi</td>
                <td>0.40 (0.00-0.43)</td>
                <td>Tier 2</td>
                <td>9</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>ERNIE X1 Turbo</td>
                <td>0.13 (0.00-0.32)</td>
                <td>Tier 3</td>
                <td>10</td>
              </tr>
              <tr valign="top">
                <td>Chinese_Expert</td>
                <td>ChatGLM</td>
                <td>0.04 (0.00-0.35)</td>
                <td>Tier 3</td>
                <td>11</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>ChatGPT</td>
                <td>0.95 (0.88-1.00)</td>
                <td>Tier 1</td>
                <td>1</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>DeepSeek</td>
                <td>0.93 (0.82-1.00)</td>
                <td>Tier 1</td>
                <td>2</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>Qwen</td>
                <td>0.90 (0.81-0.98)</td>
                <td>Tier 1</td>
                <td>3</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>Doubao</td>
                <td>0.88 (0.77-0.93)</td>
                <td>Tier 1</td>
                <td>4</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>Gemini</td>
                <td>0.71 (0.24-0.78)</td>
                <td>Tier 2</td>
                <td>5</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>Claude</td>
                <td>0.70 (0.24-0.77)</td>
                <td>Tier 2</td>
                <td>6</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>Hunyuan</td>
                <td>0.67 (0.20-0.73)</td>
                <td>Tier 2</td>
                <td>7</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>Grok</td>
                <td>0.65 (0.09-0.71)</td>
                <td>Tier 2</td>
                <td>8</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>ERNIE X1 Turbo</td>
                <td>0.14 (0.02-0.37)</td>
                <td>Tier 3</td>
                <td>9</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>Kimi</td>
                <td>0.07 (0.00-0.38)</td>
                <td>Tier 3</td>
                <td>10</td>
              </tr>
              <tr valign="top">
                <td>Chinese_No-context</td>
                <td>ChatGLM</td>
                <td>0.05 (0.00-0.39)</td>
                <td>Tier 3</td>
                <td>11</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>ChatGPT</td>
                <td>0.93 (0.76-0.94)</td>
                <td>Tier 1</td>
                <td>1</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>Qwen</td>
                <td>0.92 (0.79-1.00)</td>
                <td>Tier 1</td>
                <td>2</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>DeepSeek</td>
                <td>0.86 (0.63-0.97)</td>
                <td>Tier 1</td>
                <td>3</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>Doubao</td>
                <td>0.75 (0.36-0.83)</td>
                <td>Tier 2</td>
                <td>4</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>Hunyuan</td>
                <td>0.74 (0.33-0.84)</td>
                <td>Tier 2</td>
                <td>5</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>Claude</td>
                <td>0.70 (0.29-0.80)</td>
                <td>Tier 2</td>
                <td>6</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>Gemini</td>
                <td>0.70 (0.25-0.78)</td>
                <td>Tier 2</td>
                <td>7</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>Grok</td>
                <td>0.67 (0.17-0.75)</td>
                <td>Tier 2</td>
                <td>8</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>Kimi</td>
                <td>0.64 (0.23-0.73)</td>
                <td>Tier 2</td>
                <td>9</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>ERNIE X1 Turbo</td>
                <td>0.26 (0.00-0.35)</td>
                <td>Tier 3</td>
                <td>10</td>
              </tr>
              <tr valign="top">
                <td>English_Expert</td>
                <td>ChatGLM</td>
                <td>0.00 (0.00-0.40)</td>
                <td>Tier 3</td>
                <td>11</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>ChatGPT</td>
                <td>0.96 (0.88-1.00)</td>
                <td>Tier 1</td>
                <td>1</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>Gemini</td>
                <td>0.90 (0.85-0.98)</td>
                <td>Tier 1</td>
                <td>2</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>Qwen</td>
                <td>0.88 (0.83-0.98)</td>
                <td>Tier 1</td>
                <td>3</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>Grok</td>
                <td>0.88 (0.81-0.96)</td>
                <td>Tier 1</td>
                <td>4</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>Doubao</td>
                <td>0.85 (0.73-0.94)</td>
                <td>Tier 1</td>
                <td>5</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>Hunyuan</td>
                <td>0.78 (0.49-0.87)</td>
                <td>Tier 2</td>
                <td>6</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>Claude</td>
                <td>0.70 (0.29-0.78)</td>
                <td>Tier 2</td>
                <td>7</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>DeepSeek</td>
                <td>0.70 (0.22-0.78)</td>
                <td>Tier 2</td>
                <td>8</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>Kimi</td>
                <td>0.23 (0.02-0.37)</td>
                <td>Tier 3</td>
                <td>9</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>ERNIE X1 Turbo</td>
                <td>0.22 (0.06-0.38)</td>
                <td>Tier 3</td>
                <td>10</td>
              </tr>
              <tr valign="top">
                <td>English_No-context</td>
                <td>ChatGLM</td>
                <td>0.00 (0.00-0.40)</td>
                <td>Tier 3</td>
                <td>11</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Radar charts of multidimensional quality scores under the no-context condition. Technique for order preference by similarity to ideal solution (TOPSIS)–based radar visualization comparing the top 6 ranked models (by equal-weight TOPSIS score) on 4 quality dimensions for (A) Chinese queries (no-context) and (B) English queries (no-context). All scores were normalized to a 0 to 1 scale (accuracy divided by 100; other metrics divided by 10).</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e96587_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>TOPSIS sensitivity analysis across alternative weighting schemes. (A) TOPSIS scores for 11 models under the Chinese no-context condition, compared across 3 weighting schemes: Equal (accuracy 25%, comprehensiveness 25%, precision 25%, understandability 25%), Safety-weighted (accuracy 50%, comprehensiveness 16.7%, precision 16.7%, understandability 16.7%), and Communication-weighted (accuracy 20%, comprehensiveness 30%, precision 25%, understandability 25%). (B) Tier stability across all four languages × prompt conditions and all three weighting schemes. Stacked horizontal bars show the number of conditions × scheme combinations in which each model was classified as Tier 1 (top performers), Tier 2 (mid-tier), or Tier 3 (bottom performers).</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e96587_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Sensitivity analyses using safety-weighted (accuracy: 50% and others: 16.7% each) and communication-weighted (accuracy: 20%, comprehensiveness: 30%, precision: 25%, and understandability: 25%) schemes demonstrated robust tier stability (<xref ref-type="supplementary-material" rid="app17">Multimedia Appendix 17</xref>). ChatGPT and Qwen maintained tier 1 classification in all 12 conditions×weight−scheme combinations (100% stability). DeepSeek was classified as tier 1 in 9 (75%) of the 12 combinations, consistently dropping to tier 2 in English no-context conditions under all 3 weighting schemes (rank #7-8), independently confirming the LMM interaction findings. Both ERNIE X1 Turbo and ChatGLM were classified as tier 3 in all 12 combinations (100% stability). This stability at both extremes confirmed that the TOPSIS tier classifications were insensitive to weighting choices for the highest- and lowest-performing models.</p>
      </sec>
      <sec>
        <title>Post Hoc Error-Correction Strategies</title>
        <p>The following analyses were conducted exclusively on baseline-incorrect responses (overall baseline error rate: Chinese: 102/1650, 6.18% and English: 78/1650, 4.73%) and were presented separately from the full-benchmark prompt conditions above. Δ-accuracy values represented within-error-subset correction rates (<xref ref-type="supplementary-material" rid="app18">Multimedia Appendix 18</xref>; <xref rid="figure5" ref-type="fig">Figure 5</xref>).</p>
        <fig id="figure5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Post hoc accuracy improvement (Δ-Accuracy) following error-correction strategies. (A) Overall improvement pooled across both languages and all 11 models. (B) Comparison of Δ-Accuracy between Chinese (cyan) and English (red) queries. CoT: chain-of-thought; RAG: retrieval-augmented generation.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e96587_fig5.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>CoT prompting achieved a correction rate of 48.33% (87/180) overall (Chinese: 46/102, 45.10% and English: 41/78, 52.56%). RAG yielded moderate improvements (71/180, 39.44%), and combined CoT+RAG (69/180, 38.33%) did not produce additive effects. The error-attribution condition achieved the highest correction rate (100/180, 55.56%; Chinese: 52/102, 50.98% and English: 48/78, 61.54%); however, this condition provided models with privileged information that their prior answer was incorrect and should be interpreted as an upper-bound diagnostic test.</p>
        <p>The expert prompt was associated with an overall Δ of −53.33% (−96/180, Chinese: −33/102, −32.35% and English: −63/78, −80.77%). These negative Δ values indicated that the expert prompt introduced more new errors than it corrected. When models had very few baseline errors (eg, Grok English: 4 errors), introducing even a small number of new errors produced extremely large negative Δ values (eg, −13/4, −325%).</p>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <p>We evaluated 11 LLMs on 150 binary consumer health questions in both English and Chinese. Aggregate accuracy was high across languages (Chinese: 1548/1650, 93.82% and English: 1572/1650, 95.27%), with a precisely null language main effect. No language main effects for any individual model remained significant after FDR correction, consistent with recent cross-lingual benchmarking showing that question type, rather than language, was the primary determinant of LLM performance on the Chinese National Medical Licensing Examination [<xref ref-type="bibr" rid="ref17">17</xref>]. Similar conclusions emerged from a multilingual evaluation across German, French, and Italian medical examinations, where performance variability across models exceeded variability across languages [<xref ref-type="bibr" rid="ref18">18</xref>].</p>
      <p>The absence of a systematic language effect on accuracy across 11 models was broadly reassuring. Earlier evaluations consistently reported performance drops when medical LLMs were queried in non-English languages. Jin et al [<xref ref-type="bibr" rid="ref13">13</xref>] found that English prompts yielded substantially higher accuracy across multiple languages, and a comprehensive platform spanning 67 countries and 27 languages documented persistent, albeit narrowing, English advantages in medical examination performance [<xref ref-type="bibr" rid="ref19">19</xref>]. Our finding of near-identical English-Chinese accuracy indicated that the cross-linguistic gap had largely closed for this high-resource language pair in binary question answering, though the extent to which this generalizes to open-ended clinical tasks remained unclear.</p>
      <p>The concentration of the language interaction in DeepSeek alone pointed to model-level rather than language-level variation. DeepSeek achieved the highest Chinese understandability score (mean: 7.92) among all models tested but dropped to 7.22 in English. Previous work has reported a similar pattern. In an evaluation of prostate cancer radiotherapy questions, DeepSeek received top-quality ratings in 75.8% of Chinese responses vs ChatGPT’s 36.4%, while the relationship shifted in English [<xref ref-type="bibr" rid="ref20">20</xref>]. In bilingual neuro-oncology consultations involving ChatGPT-4o, DeepSeek-R1, and Doubao, all 3 exceeded 90% diagnostic accuracy across languages, but language-dependent differences emerged in specific clinical tasks [<xref ref-type="bibr" rid="ref21">21</xref>]. On the 2024 Chinese National Medical Licensing Examination, DeepSeek-R1 achieved 92.0% accuracy compared with GPT-4o’s 87.2% [<xref ref-type="bibr" rid="ref22">22</xref>]. In a separate evaluation using the 2021 examination, Chinese-developed models consistently outperformed OpenAI models, with ERNIE 4.5 Turbo reaching 95.3% and Qwen 3 reaching 92.5% [<xref ref-type="bibr" rid="ref23">23</xref>]. Our study extended these observations by showing that DeepSeek’s cross-linguistic limitation was specific to communicative clarity rather than factual accuracy. ChatGPT maintained balanced high performance across both languages, and Qwen demonstrated robust stability.</p>
      <p>The expert prompt reduced comprehensiveness and precision but did not affect understandability. This pattern should be interpreted with caution. The expert prompt differed from the no-context prompt in several respects beyond the expert framing itself, including prompt length and forced response structuring. Any of these design features could have contributed to the observed effect. Several models (DeepSeek, Qwen, Kimi, and Grok) showed significantly less degradation than the reference model (ChatGLM), suggesting model-specific resilience to prompt-format effects. These results were specific to this prompt design, binary scoring framework, and benchmark, and should not be taken as evidence that expert-framed prompting generally impaired LLM performance. Among post hoc correction strategies applied to baseline errors, CoT prompting achieved the highest correction rate among realistic strategies. Error attribution achieved a higher rate but should be regarded as an artificial upper bound, as it explicitly informed models of prior errors.</p>
      <p>This study has several limitations. First, binary health questions, while enabling objective scoring, did not capture complex clinical reasoning involving probabilistic assessment, treatment trade-offs, uncertainty expression, shared decision-making, or harm avoidance. The high baseline accuracy created ceiling effects that limit performance discrimination. Second, several models had built-in web search enabled during evaluation (DeepSeek R1, Kimi k1.5, Hunyuan T1, ERNIE X1 Turbo, ChatGLM 4, and GPT-4.5), which may have influenced responses through retrieval of language-specific web content. Third, ground-truth labels were used as provided by the TREC Health Misinformation Track without independent revalidation against current evidence. Fourth, results reflect model capabilities at a single time point (May 2025) and may not generalize to subsequent model versions. Fifth, RAG findings were exploratory, as Google Search results were dynamic, nonreproducible, and not the primary search engine used in mainland China. Finally, our English-Chinese comparison could not be generalized to low-resource languages where LLM performance may differ substantially.</p>
      <p>Contemporary LLMs achieved high binary accuracy on consumer health questions in both English and Chinese (&gt;93%), with no significant aggregate language effect on either accuracy or communication quality. Model-specific variation in communication quality was concentrated in a single model and single metric: DeepSeek’s English understandability decrement, which was independently confirmed by TOPSIS tier analysis. One 3-way interaction (Grok×English×expert, FDR <italic>P</italic>=.02) indicated model-specific sensitivity to combined cross-linguistic and prompt-format conditions. Cross-linguistic communication quality appeared to be a model-specific rather than universal concern, warranting targeted monitoring in individual model evaluation for multilingual health information deployment.</p>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Model specifications and access details for 11 large language models.</p>
        <media xlink:href="jmir_v28i1e96587_app1.docx" xlink:title="DOCX File , 14 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Institutional ethics exemption letter.</p>
        <media xlink:href="jmir_v28i1e96587_app2.docx" xlink:title="DOCX File , 3894 KB"/>
      </supplementary-material>
      <supplementary-material id="app3">
        <label>Multimedia Appendix 3</label>
        <p>150 questions (English + Chinese) with ground-truth labels.</p>
        <media xlink:href="jmir_v28i1e96587_app3.docx" xlink:title="DOCX File , 34 KB"/>
      </supplementary-material>
      <supplementary-material id="app4">
        <label>Multimedia Appendix 4</label>
        <p>Detailed prompts both in Chinese and English.</p>
        <media xlink:href="jmir_v28i1e96587_app4.docx" xlink:title="DOCX File , 17 KB"/>
      </supplementary-material>
      <supplementary-material id="app5">
        <label>Multimedia Appendix 5</label>
        <p>Full scoring rubric with anchors and examples (1-3/4-6/7-9/10 anchors for each metric).</p>
        <media xlink:href="jmir_v28i1e96587_app5.docx" xlink:title="DOCX File , 21 KB"/>
      </supplementary-material>
      <supplementary-material id="app6">
        <label>Multimedia Appendix 6</label>
        <p>Representative retrieval-augmented generation search logs.</p>
        <media xlink:href="jmir_v28i1e96587_app6.docx" xlink:title="DOCX File , 98 KB"/>
      </supplementary-material>
      <supplementary-material id="app7">
        <label>Multimedia Appendix 7</label>
        <p>TRIPOD-LLM Reporting Checklist.</p>
        <media xlink:href="jmir_v28i1e96587_app7.docx" xlink:title="DOCX File , 25 KB"/>
      </supplementary-material>
      <supplementary-material id="app8">
        <label>Multimedia Appendix 8</label>
        <p>Inter-rater reliability:Cohen κ and weighted κ values.</p>
        <media xlink:href="jmir_v28i1e96587_app8.docx" xlink:title="DOCX File , 13 KB"/>
      </supplementary-material>
      <supplementary-material id="app9">
        <label>Multimedia Appendix 9</label>
        <p>Accuracy by dataset-year subset for 11 large language models under no-context prompting. Grouped bar charts showing model-level accuracy (%) stratified by Text Retrieval Conference Health Misinformation Track dataset year: 2019, 2021, and 2022. (A) Chinese-language input. (B) English-language input.</p>
        <media xlink:href="jmir_v28i1e96587_app9.png" xlink:title="PNG File , 237 KB"/>
      </supplementary-material>
      <supplementary-material id="app10">
        <label>Multimedia Appendix 10</label>
        <p>Distribution of communication quality scores by language for 11 large language models. Split violin plots showing the distribution of scores under the no-context condition, with Chinese responses (blue, left half) and English responses (red, right half) displayed as mirrored density distributions. Inner box plots display the median and interquartile range. (A) Comprehensiveness. (B) Precision. (C) Understandability. Each violin represents responses across 150 health questions scored on a 1-10 scale by 2 independent bilingual clinical physicians.</p>
        <media xlink:href="jmir_v28i1e96587_app10.png" xlink:title="PNG File , 112 KB"/>
      </supplementary-material>
      <supplementary-material id="app11">
        <label>Multimedia Appendix 11</label>
        <p>Quality metric descriptive statistics by model, language, and prompt condition.</p>
        <media xlink:href="jmir_v28i1e96587_app11.docx" xlink:title="DOCX File , 18 KB"/>
      </supplementary-material>
      <supplementary-material id="app12">
        <label>Multimedia Appendix 12</label>
        <p>Three-way interaction effects (model×language×prompt) on 4 metrics. Forest plots showing coefficient estimates (dots) with 95% CIs (horizontal lines) for 3-way interaction terms (model×english×expert prompt) from (A) generalized estimating equations for accuracy and (B-D) linear mixed models for comprehensiveness, understandability, and precision, respectively. Each panel displays the 10 model×english×expert prompt interaction terms, with ChatGLM as the reference model, Chinese as the reference language, and no-context as the reference prompt. GEE: generalized estimating equation; LMM: linear mixed model; P2: expert prompt condition.</p>
        <media xlink:href="jmir_v28i1e96587_app12.png" xlink:title="PNG File , 229 KB"/>
      </supplementary-material>
      <supplementary-material id="app13">
        <label>Multimedia Appendix 13</label>
        <p>Full generalized estimating equation coefficients for binary accuracy with false discovery rate correction.</p>
        <media xlink:href="jmir_v28i1e96587_app13.docx" xlink:title="DOCX File , 20 KB"/>
      </supplementary-material>
      <supplementary-material id="app14">
        <label>Multimedia Appendix 14</label>
        <p>Full linear mixed model coefficients for quality metrics with false discovery rate correction.</p>
        <media xlink:href="jmir_v28i1e96587_app14.docx" xlink:title="DOCX File , 40 KB"/>
      </supplementary-material>
      <supplementary-material id="app15">
        <label>Multimedia Appendix 15</label>
        <p>Three-way interaction effects (model×language×prompt) across all metrics.</p>
        <media xlink:href="jmir_v28i1e96587_app15.docx" xlink:title="DOCX File , 19 KB"/>
      </supplementary-material>
      <supplementary-material id="app16">
        <label>Multimedia Appendix 16</label>
        <p>Two-by-two composite of 4 forest plot panels: (A) accuracy from a generalized estimating equation (GEE) with binomial link, (B) comprehensiveness, (C) understandability, and (D) precision from linear mixed-effects models (LMMs).</p>
        <media xlink:href="jmir_v28i1e96587_app16.png" xlink:title="PNG File , 194 KB"/>
      </supplementary-material>
      <supplementary-material id="app17">
        <label>Multimedia Appendix 17</label>
        <p>Technique for order of preference by similarity to ideal solution (TOPSIS) sensitivity analysis: composite scores, tiers, and rankings under 3 weighting schemes.</p>
        <media xlink:href="jmir_v28i1e96587_app17.docx" xlink:title="DOCX File , 34 KB"/>
      </supplementary-material>
      <supplementary-material id="app18">
        <label>Multimedia Appendix 18</label>
        <p>Post hoc error correction rates (Δ-accuracy) by model and language, with pooled counts.</p>
        <media xlink:href="jmir_v28i1e96587_app18.docx" xlink:title="DOCX File , 20 KB"/>
      </supplementary-material>
      <supplementary-material id="app19">
        <label>Multimedia Appendix 19</label>
        <p>Detailed prompts and complete conversation logs of Claude Opus 4.6 (Anthropic).</p>
        <media xlink:href="jmir_v28i1e96587_app19.docx" xlink:title="DOCX File , 15 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">CoT</term>
          <def>
            <p>chain-of-thought</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">FDR</term>
          <def>
            <p>false discovery rate</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">GEE</term>
          <def>
            <p>generalized estimating equation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">LMM</term>
          <def>
            <p>linear mixed model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">RAG</term>
          <def>
            <p>retrieval-augmented generation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">TOPSIS</term>
          <def>
            <p>technique for order of preference by similarity to ideal solution</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">TREC</term>
          <def>
            <p>Text Retrieval Conference</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">TRIPOD-LLM</term>
          <def>
            <p>Transparent Reporting of a Multivariable Model for Individual Prognosis or Diagnosis–Large Language Model</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors would like to thank the TREC Text Retrieval Conference Health Misinformation Track organizers for making the question datasets publicly available. Claude Opus 4.6 (Anthropic) was used solely for language polishing and stylistic refinement of the manuscript. It was not involved in study conceptualization, data collection, analysis, interpretation of results, or generation of intellectual content. The authors take full responsibility for the accuracy and integrity of the work. Complete conversation logs are available as <xref ref-type="supplementary-material" rid="app19">Multimedia Appendix 19</xref>.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>All data analyzed in this study are publicly available through the Text Retrieval Conference Health Misinformation Track.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>This research was supported by grants from the National High Level Hospital Clinical Research Funding (2025-PUMCH-A-087), Beijing Natural Science Foundation (L256086), the National Key R&amp;D Program of China (2024YFA1307604), the National Natural Science Foundation of China Grants (82472348), National Science and Technology Major Project (2025ZD0551304), and Peking Union Medical College Hospital Talent Cultivation Program (UBJ06102).</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>ZW, HX, and FF contributed to the study design, performed the experiments, and drafted the manuscript. RC, TS, and SW were involved in data acquisition and statistical analysis. SZ and YL provided supervision and critically revised the manuscript for intellectual content. All authors approved the final version. YL is the guarantor of this work and had full access to all study data.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bedi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Orr-Ewing</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Dash</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Koyejo</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Callahan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Fries</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Wornow</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Swaminathan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lehmann</surname>
              <given-names>LS</given-names>
            </name>
            <name name-style="western">
              <surname>Hong</surname>
              <given-names>HJ</given-names>
            </name>
            <name name-style="western">
              <surname>Kashyap</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Chaurasia</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NR</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tazbaz</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Milstein</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pfeffer</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NH</given-names>
            </name>
          </person-group>
          <article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title>
          <source>JAMA</source>
          <year>2025</year>
          <month>01</month>
          <day>28</day>
          <volume>333</volume>
          <issue>4</issue>
          <fpage>319</fpage>
          <lpage>28</lpage>
          <pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id>
          <pub-id pub-id-type="medline">39405325</pub-id>
          <pub-id pub-id-type="pii">2825147</pub-id>
          <pub-id pub-id-type="pmcid">PMC11480901</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Han</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Large language models in biomedicine and healthcare</article-title>
          <source>NPJ Artif Intell</source>
          <year>2025</year>
          <month>12</month>
          <day>01</day>
          <volume>1</volume>
          <fpage>44</fpage>
          <pub-id pub-id-type="doi">10.1038/s44387-025-00047-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Fernández-Pichel</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pichel</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Losada</surname>
              <given-names>DE</given-names>
            </name>
          </person-group>
          <article-title>Evaluating search engines and large language models for answering health questions</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>03</month>
          <day>10</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>153</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01546-w"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01546-w</pub-id>
          <pub-id pub-id-type="medline">40065094</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01546-w</pub-id>
          <pub-id pub-id-type="pmcid">PMC11894092</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Asgari</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Montaña-Brown</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Dubois</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Khalil</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Balloch</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yeung</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Pimenta</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>05</month>
          <day>13</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>274</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01670-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01670-7</pub-id>
          <pub-id pub-id-type="medline">40360677</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01670-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC12075489</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Granstedt</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kc</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Deshpande</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Garcia</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Badano</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Hallucinations in medical devices</article-title>
          <source>Artif Intell Life Sci</source>
          <year>2025</year>
          <month>12</month>
          <volume>8</volume>
          <fpage>100145</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1016/j.ailsci.2025.100145"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.ailsci.2025.100145</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sayres</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wulczyn</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Amin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hou</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>SR</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Neal</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Rashid</surname>
              <given-names>QM</given-names>
            </name>
            <name name-style="western">
              <surname>Schaekermann</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dash</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NH</given-names>
            </name>
            <name name-style="western">
              <surname>Lachgar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Prakash</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Green</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Dominowska</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Tomašev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>JK</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>DR</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Toward expert-level medical question answering with large language models</article-title>
          <source>Nat Med</source>
          <year>2025</year>
          <month>03</month>
          <volume>31</volume>
          <issue>3</issue>
          <fpage>943</fpage>
          <lpage>50</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id>
          <pub-id pub-id-type="medline">39779926</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-03423-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11922739</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tanwani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Seneviratne</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gamble</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Babiker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schärli</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhery</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tomasev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>08</month>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>80</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Roustan</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Bastardot</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>The clinicians' guide to large language models: a general perspective with a focus on hallucinations</article-title>
          <source>Interact J Med Res</source>
          <year>2025</year>
          <month>01</month>
          <day>28</day>
          <volume>14</volume>
          <fpage>e59823</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.i-jmr.org/2025//e59823/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/59823</pub-id>
          <pub-id pub-id-type="medline">39874574</pub-id>
          <pub-id pub-id-type="pii">v14i1e59823</pub-id>
          <pub-id pub-id-type="pmcid">PMC11815294</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ong</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Ning</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Bitterman</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Tham</surname>
              <given-names>YC</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Jiménez de Tavárez</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Mateen</surname>
              <given-names>BA</given-names>
            </name>
            <name name-style="western">
              <surname>Amissah-Arthur</surname>
              <given-names>KN</given-names>
            </name>
            <name name-style="western">
              <surname>Sheng</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>IB</given-names>
            </name>
            <name name-style="western">
              <surname>Hong</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>LT</given-names>
            </name>
            <name name-style="western">
              <surname>Goldstein</surname>
              <given-names>BA</given-names>
            </name>
            <name name-style="western">
              <surname>Le</surname>
              <given-names>PV</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>HK</given-names>
            </name>
            <name name-style="western">
              <surname>Ong</surname>
              <given-names>ME</given-names>
            </name>
            <name name-style="western">
              <surname>Wagner</surname>
              <given-names>SK</given-names>
            </name>
            <name name-style="western">
              <surname>Denniston</surname>
              <given-names>AK</given-names>
            </name>
            <name name-style="western">
              <surname>Keane</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Car</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chapman</surname>
              <given-names>WW</given-names>
            </name>
            <name name-style="western">
              <surname>Moons</surname>
              <given-names>KG</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>TY</given-names>
            </name>
            <name name-style="western">
              <surname>Topol</surname>
              <given-names>EJ</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>N</given-names>
            </name>
          </person-group>
          <article-title>Large language models in global health</article-title>
          <source>Nat Health</source>
          <year>2026</year>
          <month>01</month>
          <day>15</day>
          <volume>1</volume>
          <fpage>35</fpage>
          <lpage>47</lpage>
          <pub-id pub-id-type="doi">10.1038/s44360-025-00024-7</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gallifant</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Afshar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ameen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Aphinyanaphongs</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cacciamani</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Dligach</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Daneshjou</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Fernandes</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hansen</surname>
              <given-names>LH</given-names>
            </name>
            <name name-style="western">
              <surname>Landman</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lehmann</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>McCoy</surname>
              <given-names>LG</given-names>
            </name>
            <name name-style="western">
              <surname>Miller</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Moreno</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Munch</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Restrepo</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Savova</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Umeton</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Gichoya</surname>
              <given-names>JW</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Moons</surname>
              <given-names>KG</given-names>
            </name>
            <name name-style="western">
              <surname>Celi</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Bitterman</surname>
              <given-names>DS</given-names>
            </name>
          </person-group>
          <article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title>
          <source>Nat Med</source>
          <year>2025</year>
          <month>01</month>
          <volume>31</volume>
          <issue>1</issue>
          <fpage>60</fpage>
          <lpage>9</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id>
          <pub-id pub-id-type="medline">39779929</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-03425-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC12104976</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Thirunavukarasu</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Elangovan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gutierrez</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>TF</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DS</given-names>
            </name>
          </person-group>
          <article-title>Large language models in medicine</article-title>
          <source>Nat Med</source>
          <year>2023</year>
          <month>08</month>
          <volume>29</volume>
          <issue>8</issue>
          <fpage>1930</fpage>
          <lpage>40</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id>
          <pub-id pub-id-type="medline">37460753</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-023-02448-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Bi</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>ZF</given-names>
            </name>
            <name name-style="western">
              <surname>Gou</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Shao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Xue</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Feng</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Deng</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Ruan</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Dai</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Ji</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Dai</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hao</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Ding</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Qu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yuan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Qiu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Cai</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Ni</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Dong</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>You</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Guan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Xia</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Tian</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Du</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Ge</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Pan</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>RL</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ye</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pan</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Yun</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Pei</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Zeng</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Xiao</surname>
              <given-names>WL</given-names>
            </name>
            <name name-style="western">
              <surname>An</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Nie</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Su</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>XQ</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Shan</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>YK</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>YQ</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>YX</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Shi</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xiong</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Piao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Gong</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xiong</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>You</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>YX</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zha</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>ZZ</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Sha</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Fu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Hao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yan</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Gu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Pan</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning</article-title>
          <source>Nature</source>
          <year>2025</year>
          <month>09</month>
          <volume>645</volume>
          <issue>8081</issue>
          <fpage>633</fpage>
          <lpage>8</lpage>
          <pub-id pub-id-type="doi">10.1038/s41586-025-09422-z</pub-id>
          <pub-id pub-id-type="medline">40962978</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-025-09422-z</pub-id>
          <pub-id pub-id-type="pmcid">PMC12443585</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chandra</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Verma</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>De Choudhury</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kumar</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Better to ask in English: cross-lingual evaluation of large language models for healthcare queries</article-title>
          <source>Proceedings of the ACM Web Conference 2024</source>
          <year>2024</year>
          <conf-name>WWW '24</conf-name>
          <conf-date>May 13-17, 2024</conf-date>
          <conf-loc>Singapore</conf-loc>
          <pub-id pub-id-type="doi">10.1145/3589334.3645643</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Workum</surname>
              <given-names>JD</given-names>
            </name>
            <name name-style="western">
              <surname>van de Sande</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Gommers</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>van Genderen</surname>
              <given-names>ME</given-names>
            </name>
          </person-group>
          <article-title>Bridging the gap: a practical step-by-step approach to warrant safe implementation of large language models in healthcare</article-title>
          <source>Front Artif Intell</source>
          <year>2025</year>
          <month>01</month>
          <day>27</day>
          <volume>8</volume>
          <fpage>1504805</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/frai.2025.1504805"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/frai.2025.1504805</pub-id>
          <pub-id pub-id-type="medline">39931218</pub-id>
          <pub-id pub-id-type="pmcid">PMC11808533</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Jeong</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Alhamoud</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Mun</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Grau</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Jung</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gameiro</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Fan</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Yoon</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yoon</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Sap</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Tsvetkov</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>McDuff</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Tulebaev</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Breazeal</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Medical hallucination in foundation models and their impact on healthcare</article-title>
          <source>medRxiv. Preprint posted online on March 03, 2025</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.medrxiv.org/content/10.1101/2025.02.28.25323115v1.full"/>
          </comment>
          <pub-id pub-id-type="doi">10.1101/2025.02.28.25323115</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hasnain</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Aurangzeb</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Alhussein</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ghani</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Mahmood</surname>
              <given-names>MH</given-names>
            </name>
          </person-group>
          <article-title>AI in conjunctivitis research: assessing ChatGPT and DeepSeek for etiology, intervention, and citation integrity via hallucination rate analysis</article-title>
          <source>Front Artif Intell</source>
          <year>2025</year>
          <month>8</month>
          <day>20</day>
          <volume>8</volume>
          <fpage>1579375</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.3389/frai.2025.1579375"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/frai.2025.1579375</pub-id>
          <pub-id pub-id-type="medline">40910118</pub-id>
          <pub-id pub-id-type="pmcid">PMC12405273</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Performance benchmarking of LLMs on Chinese national medical licensing education: cross-lingual and question-type effects</article-title>
          <source>PLoS One</source>
          <year>2026</year>
          <month>04</month>
          <day>08</day>
          <volume>21</volume>
          <issue>4</issue>
          <fpage>e0346518</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pone.0346518"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0346518</pub-id>
          <pub-id pub-id-type="medline">41950260</pub-id>
          <pub-id pub-id-type="pii">PONE-D-25-45754</pub-id>
          <pub-id pub-id-type="pmcid">PMC13061252</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Strasser</surname>
              <given-names>LM</given-names>
            </name>
            <name name-style="western">
              <surname>Anschuetz</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Dennstädt</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hastings</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Performance evaluation of large language models in multilingual medical multiple-choice questions: mixed methods study</article-title>
          <source>JMIR Med Educ</source>
          <year>2026</year>
          <month>03</month>
          <day>05</day>
          <volume>12</volume>
          <fpage>e81399</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mededu.jmir.org/2026//e81399/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/81399</pub-id>
          <pub-id pub-id-type="medline">41813244</pub-id>
          <pub-id pub-id-type="pii">v12i1e81399</pub-id>
          <pub-id pub-id-type="pmcid">PMC12978932</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zong</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Cha</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Feng</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Large language models in worldwide medical exams: platform development and comprehensive analysis</article-title>
          <source>J Med Internet Res</source>
          <year>2024</year>
          <month>12</month>
          <day>27</day>
          <volume>26</volume>
          <fpage>e66114</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2024//e66114/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/66114</pub-id>
          <pub-id pub-id-type="medline">39729356</pub-id>
          <pub-id pub-id-type="pii">v26i1e66114</pub-id>
          <pub-id pub-id-type="pmcid">PMC11724220</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>PW</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>JW</given-names>
            </name>
            <name name-style="western">
              <surname>Xie</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>JW</given-names>
            </name>
            <name name-style="western">
              <surname>Huo</surname>
              <given-names>XY</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>ZL</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>ZC</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>SQ</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>MQ</given-names>
            </name>
          </person-group>
          <article-title>DeepSeek vs ChatGPT: a comparison study of their performance in answering prostate cancer radiotherapy questions in multiple languages</article-title>
          <source>Am J Clin Exp Urol</source>
          <year>2025</year>
          <month>04</month>
          <day>25</day>
          <volume>13</volume>
          <issue>2</issue>
          <fpage>176</fpage>
          <lpage>85</lpage>
          <pub-id pub-id-type="doi">10.62347/UIAP7979</pub-id>
          <pub-id pub-id-type="medline">40400997</pub-id>
          <pub-id pub-id-type="pmcid">PMC12089221</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Pan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tian</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Cai</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wan</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Fang</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Clinical feasibility of AI doctors: evaluating the replacement potential of large language models in outpatient settings for central nervous system tumors</article-title>
          <source>Int J Med Inform</source>
          <year>2025</year>
          <month>11</month>
          <volume>203</volume>
          <fpage>106013</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1386-5056(25)00230-8"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.106013</pub-id>
          <pub-id pub-id-type="medline">40554367</pub-id>
          <pub-id pub-id-type="pii">S1386-5056(25)00230-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Long</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Cao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>He</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of DeepSeek-R1 and ChatGPT-4o on the Chinese National Medical Licensing Examination: a multi-year comparative study</article-title>
          <source>Sci Rep</source>
          <year>2026</year>
          <month>01</month>
          <day>12</day>
          <volume>16</volume>
          <issue>1</issue>
          <fpage>2237</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-025-31874-6"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-025-31874-6</pub-id>
          <pub-id pub-id-type="medline">41526606</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-025-31874-6</pub-id>
          <pub-id pub-id-type="pmcid">PMC12816717</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Qin</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Performance of DeepSeek-R1 and ChatGPT-4o on the Chinese National Medical Licensing Examination: a comparative study</article-title>
          <source>J Med Syst</source>
          <year>2025</year>
          <month>06</month>
          <day>03</day>
          <volume>49</volume>
          <issue>1</issue>
          <fpage>74</fpage>
          <pub-id pub-id-type="doi">10.1007/s10916-025-02213-z</pub-id>
          <pub-id pub-id-type="medline">40459679</pub-id>
          <pub-id pub-id-type="pii">10.1007/s10916-025-02213-z</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
