<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e89963</article-id><article-id pub-id-type="doi">10.2196/89963</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Stepwise Diagnostic Evaluation of Chinese Large Language Models: Comparative Study of Common and Rare Diseases</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Wang</surname><given-names>Jiayi</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Yang</surname><given-names>Jiao</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Guo</surname><given-names>Rui</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Department of Health Management and Policy, School of Public Health, Capital Medical University</institution><addr-line>No. 10 Xitoutiao, Youanmenwai, Fengtai District</addr-line><addr-line>Beijing</addr-line><addr-line>Beijing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Lv</surname><given-names>Huasheng</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Soleimani</surname><given-names>Mohammad</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Luo</surname><given-names>Peng</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Rui Guo, PhD, Department of Health Management and Policy, School of Public Health, Capital Medical University, No. 10 Xitoutiao, Youanmenwai, Fengtai District, Beijing, Beijing, 100069, China, 86 010 83911303; <email>guorui@ccmu.edu.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>6</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e89963</elocation-id><history><date date-type="received"><day>22</day><month>12</month><year>2025</year></date><date date-type="rev-recd"><day>02</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>03</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jiayi Wang, Jiao Yang, Rui Guo. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 6.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e89963"/><abstract><sec><title>Background</title><p>Large language models (LLMs) are increasingly applied in clinical decision support, yet their diagnostic performance in Chinese-language settings and under realistic clinical workflows remains unclear. In particular, how LLMs perform across diseases with different prevalence and under stepwise diagnostic processes has not been well characterized.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the diagnostic capabilities of LLMs for common diseases and rare diseases using clinical vignettes within a hypothetico-deductive framework and to identify their potential and limitations for clinical diagnosis.</p></sec><sec sec-type="methods"><title>Methods</title><p>We evaluated 4 Chinese LLMs (Doubao 1.5, DeepSeek-V3, Kimi K1.5, and Leftdoctor GPT 3.5) using 56 clinical cases (28 chronic obstructive pulmonary disease [COPD], and 28 relapsing polychondritis [RP]) sourced from the China Clinical Case Results Database (March 31-April 14, 2025). Patient information was provided incrementally, starting with the initial medical history, followed by physical examination, and laboratory results. Evaluation metrics included top-3 accuracy (RTop3D), top-1 accuracy (RTopD), final diagnostic accuracy (RFA), and mean reciprocal rank (MRR). Statistical analysis was performed using generalized estimating equations (GEE), Friedman tests, and Wilcoxon signed-rank tests with Bonferroni correction. In addition, a qualitative analysis was conducted to characterize recurrent patterns of diagnostic errors.</p></sec><sec sec-type="results"><title>Results</title><p>LLMs demonstrated significantly higher diagnostic accuracy for COPD compared to RP across all metrics (<italic>P</italic>&#x003C;.001). Diagnostic accuracy improved after additional clinical information was provided, with the improvement mainly observed in RP cases. In RP, diagnostic accuracy increased from 32.14% to 71.43% for DeepSeek and from 35.71% to 78.57% for Doubao, whereas COPD accuracy remained consistently high across all diagnostic stages (82.14%&#x2010;92.86%). For COPD, ranking performance was high and comparable among all models (MRR range: 0.82&#x2010;0.89; <italic>P</italic>=.71). In RP, diagnostic performance differed significantly among models (MRR range: 0.10&#x2010;0.39; <italic>P</italic>&#x003C;.001). Qualitative analysis showed that COPD errors were mainly related to a failure to recognize specific features, whereas RP errors involved more diverse patterns, particularly the neglect of negative evidence and the failure to recognize specific features.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Chinese LLMs demonstrated relatively strong diagnostic performance for common diseases such as COPD, but lower and less stable performance for rare diseases such as RP. Additional clinical information improved diagnostic accuracy primarily in RP cases, although differences between models remained evident under diagnostically complex conditions. Error patterns in RP cases suggest that current LLMs remain limited in their ability to integrate complex clinical information and exclusionary findings. Careful evaluation and appropriate clinical oversight remain important for their application in clinical practice.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>clinical decision-making</kwd><kwd>diagnostic accuracy</kwd><kwd>decision support systems</kwd><kwd>artificial intelligence</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Recent advances in large language models (LLMs) have demonstrated their transformative potential across diverse health care domains. These applications range from automating medical documentation and patient education to providing personalized health consultation and drug discovery [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref5">5</xref>]. Despite this broad use, AI-driven diagnostic assistance remains one of the most promising and impactful scenarios in clinical practice [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. Given the critical and complex nature of clinical decision-making, exceptionally high accuracy is required. However, diagnostic reasoning in real-world practice often involves uncertainty, incomplete information, and iterative hypothesis refinement [<xref ref-type="bibr" rid="ref10">10</xref>]. In this context, the application of LLMs raises concerns not only about &#x201C;hallucinations&#x201D; but also about their reliability and stability in supporting complex clinical reasoning processes [<xref ref-type="bibr" rid="ref11">11</xref>]. Therefore, their diagnostic performance must be carefully and systematically evaluated before being considered for clinical use.</p><p>China&#x2019;s LLM development is progressing rapidly with strong national support. However, most existing studies focus on evaluating LLMs&#x2019; diagnostic capabilities in English [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref16">16</xref>]. Given that some research indicates ChatGPT performs better with English input compared to Chinese, and that Chinese LLMs leverage large Chinese corpora and training data representative of the Chinese population, evaluating their diagnostic ability in the Chinese context is essential for their application in Chinese clinical settings. This necessity is further underscored by the unique challenges of Chinese medical natural language processing (NLP), where models must navigate highly unstructured clinical notes characterized by complex syntactic structures and nonstandardized medical abbreviations. Although some studies have explored Chinese LLMs&#x2019; diagnostic ability in Chinese contexts, research in this area remains limited [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Early assessments of LLMs&#x2019; diagnostic competence often used multiple choice questions (MCQs), which may not accurately reflect real-world clinical performance due to their structured nature. Recent studies have shifted toward clinical vignette-based evaluations that better mimic actual clinical scenarios [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. Most of these evaluations provide LLMs with complete patient data simultaneously. However, real clinical reasoning typically follows a hypothetico-deductive process, where clinicians generate an initial differential diagnosis (DDx) from limited information, then iteratively refine these hypotheses with additional data. Assessment methods simulating this incremental information provision have demonstrated reduced diagnostic accuracy in LLMs compared to approaches presenting all data at once.</p><p>The diagnostic ability of LLMs is also highly dependent on specific disease domains and task types. While previous studies have explored LLMs&#x2019; diagnostic abilities for common and rare diseases, most have been limited to single disease categories, and comparative analyses across multiple disease types are scarce. Consequently, the understanding of the variability in LLMs&#x2019; diagnostic performance across diverse disease groups remains limited [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref27">27</xref>].</p><p>In addition to the limitations discussed above, a further methodological gap exists in the current literature. Most prior studies evaluating the diagnostic performance of LLMs rely on various forms of model adaptation, including prompt engineering, few-shot learning, or task-specific fine-tuning [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. While these approaches may enhance performance, they confound the assessment of the model&#x2019;s intrinsic diagnostic capability, making it difficult to disentangle the contribution of the underlying model from that of external optimization strategies. Establishing the baseline performance of unmodified foundation models is particularly important in the context of health care. Before LLMs can be deployed in real-world clinical settings, models intended for health care implementation must be demonstrated to be accurate, reliable, and safe for use in patient care. Without a clear understanding of their intrinsic capabilities under minimal intervention, it is challenging to determine whether observed performance reflects genuine reasoning ability or artifacts of task-specific optimization.</p><p>In summary, this study used a step-by-step approach based on the hypothetico-deductive model to comparatively assess the zero-shot diagnostic reasoning capabilities of Chinese LLMs for common and rare diseases in simulated real clinical scenarios. All models were evaluated in their original form under consistent conditions, allowing for a more direct comparison of their diagnostic performance. This design helps to provide a clearer view of how LLMs perform in clinically relevant settings and may offer useful insights for their potential application in practice.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>This study was conducted and reported in accordance with the Standards for Reporting Diagnostic Accuracy Studies (STARD) guidelines to ensure transparent and complete reporting of diagnostic accuracy.</p></sec><sec id="s2-2"><title>Selection of Disease Types</title><p>This study selected representative disease models to evaluate LLM performance across different diagnostic reasoning profiles.</p><p>Chronic obstructive pulmonary disease (COPD) was selected to represent common diseases due to its significant public health burden and high incidence. COPD is a highly prevalent internal medicine condition, affecting approximately 380 million people globally and causing over 3 million deaths annually, making it the third leading cause of mortality worldwide. In China, the overall prevalence of COPD among individuals aged 40 and above is 8.6% [<xref ref-type="bibr" rid="ref30">30</xref>]. Beyond its prevalence, COPD follows well-defined diagnostic criteria, making it a suitable model for assessing LLMs&#x2019; ability to recognize high-frequency clinical patterns.</p><p>Relapsing polychondritis (RP) was selected to represent rare diseases because of its high disability rate, frequent diagnostic delays, and misdiagnoses in clinical practice, which place a substantial economic burden on both patients and society. RP is an autoimmune disease with an unclear pathological mechanism, affecting multiple organ systems. Consequently, RP was selected as our rare disease case due to these clinical challenges and its significant impact on patients. RP was specifically chosen because its diagnosis requires synthesizing heterogeneous clinical signals across multiple organ systems, making it an appropriate model for testing models&#x2019; capacity for multisystem information integration rather than simple pattern matching [<xref ref-type="bibr" rid="ref31">31</xref>].</p><p>The selection of these 2 diseases allowed for a controlled comparison between localized, pattern-consistent conditions and systemic, complex scenarios. Although COPD and RP involve different physiological systems, all cases were developed using a consistent clinical framework, including the chief complaint, history of present illness, physical examination, and key investigations. This approach was intended to keep the diagnostic process comparable across cases by focusing on how clinical information is integrated during reasoning.</p><p>The models evaluated in this study are general-purpose LLMs trained on diverse medical corpora rather than specialty-specific datasets. Evaluating their performance across diseases with distinct diagnostic structures, therefore, reflects their baseline reasoning capabilities and reduces the likelihood that results are driven by domain-specific familiarity.</p></sec><sec id="s2-3"><title>Case Development</title><p>We sourced cases from the China Clinical Case Results Database (CMCR), a national large-scale clinical case results publishing platform funded by the Chinese Association for Science and Technology and constructed by the Journal of the Chinese Medical Association. The database contains peer-reviewed clinical cases of high professional standard (eg, COPD diagnosed per Global Initiative for Chronic Obstructive Lung Disease [GOLD] criteria and RP per McAdam criteria) and provides established reference diagnoses.</p><p>Cases of COPD and RP were retrieved between March 31 and April 14, 2025. To ensure suitability for evaluating LLM&#x2019;s diagnostic performance, cases were selected based on the clarity and clinical consistency of their primary diagnoses. The final diagnosis was required to be either COPD or RP, and cases had to include detailed medical history, physical examination, laboratory tests, and other supporting diagnostic information. Cases with significant comorbidities that could confound diagnostic interpretation were excluded (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><p>To leverage the advantages of LLMs in processing unstructured data, we specifically adapted these clinician-written summaries into a format approximating the authentic language and descriptive style of real-world patients. This involved omitting nonessential sections (eg, titles, treatment discussions, and accompanying tables and videos) and naturalizing the language. Specifically, overly specialized medical expressions were modified to approximate natural patient language, as real-world patients seldom use the professional jargon found in standardized databases (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In addition, this adaptation process also aimed to reduce the likelihood that cases could be directly recognized or matched to memorized training data, thereby mitigating potential bias related to data contamination.</p><p>This prespecified and well-defined reference standard reduced potential diagnostic ambiguity during the evaluation phase and enabled consistent comparison of LLM-generated outputs against the ground truth. It also provided a structured basis for applying a rule-based semantic matching approach in the outcome assessment. Furthermore, this standardized case construction ensured that diagnostic reasoning was primarily driven by structured clinical information rather than domain-specific knowledge alone, thereby mitigating potential confounding arising from differences in medical specialties.</p><p>As this study was based on retrospective clinical vignettes and did not involve direct patient intervention or participant recruitment, no adverse events occurred during the research process. Baseline demographic characteristics, including patient age and sex, were extracted alongside the clinical information for each case.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Case selection flowchart. CMCR: China Clinical Case Results Database; COPD: chronic obstructive pulmonary disease; RP: relapsing polychondritis.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89963_fig01.png"/></fig></sec><sec id="s2-4"><title>LLM Selection</title><p>We selected 4 Chinese LLMs for evaluation, including 3 general LLMs and one health care-specific LLM. All LLMs are open-source and independently developed in China. The general LLMs selected were Doubao (Doubao 1.5), DeepSeek (DeepSeek-V3), and Kimi (Kimi K1.5), which ranked as the top 3 AI products in terms of Chinese active users in February 2025. The health care&#x2013;specific model, Leftdoctor GPT (Leftdoctor GPT 3.5), was selected from the &#x201C;2024 China Healthcare LLMs Top 30&#x201D; list released on November 5, 2024, by the Chinese Academy of Sciences ('China Internet Week&#x2019;) and the Center for Informatization Study. All models were accessed through their publicly available interfaces and evaluated under consistent conditions without task-specific adaptation. Models designed for specialized applications, including patient services, medical image analysis, scientific research and innovation, hospital management, and medical record writing, were excluded to ensure comparability across general diagnostic tasks.</p></sec><sec id="s2-5"><title>LLM Prompts</title><p>To better simulate clinical practice, we provided task prompts to LLMs, requiring them to generate a differential diagnosis based on initial information and then provide a final diagnosis after receiving examination results (<xref ref-type="fig" rid="figure2">Figure 2</xref>). As the study was conducted using Chinese LLMs, the original prompts were administered in Chinese (the complete Chinese versions are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The English translations of the prompts were:</p><disp-quote><p>You are a physician and you will:</p><list list-type="order"><list-item><p>Based on the medical history I have provided to you, form 3 differential diagnoses in order of diagnostic likelihood of prioritization, each of which can only be one disease.</p></list-item><list-item><p>After you have given your differential diagnoses, I will provide you with reports of further tests. Please give a final diagnosis based on your differential diagnosis, combined with the examination results.</p></list-item></list></disp-quote><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Clinical scenario interaction examples of simulated human-computer interaction under clinical diagnosis and hypothesis deduction processes. COPD: chronic obstructive pulmonary disease.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89963_fig02.png"/></fig></sec><sec id="s2-6"><title>Evaluation Protocol</title><p>The evaluation was framed by the deterministic reference diagnoses retrieved from the CMCR database. Since these gold-standard diagnoses were prespecified and verified, the assessment was implemented as a rule-based semantic matching protocol. To maintain consistency, we established predefined synonym boundaries referencing standardized medical vocabularies, such as <italic>ICD-10 (International Classification of Diseases, Tenth Revision)</italic> or MeSH, in alignment with established medical AI benchmarks. Within this framework, a model-generated diagnosis was recorded as correct if it aligned with the ground truth or its recognized clinical synonyms. The assessment was performed independently by 2 researchers (JYW and XNL) who were blinded to the identity of the LLMs generating the responses, with any discrepancies adjudicated by a third senior evaluator (JY) to reach a final consensus. This structured scoring process yielded a high level of interrater agreement (Cohen kappa=.950).</p></sec><sec id="s2-7"><title>Outcome Indicators</title><p>The diagnostic performance of the LLMs was evaluated using three primary outcome indicators: the rate of final diagnosis within the top 3 DDx list (RTop3D), the rate of final diagnosis as top diagnosis (RTopD), and the rate of final diagnostic accuracy (RFA). RTopD and RTop3D were used to assess the accuracy of the models&#x2019; initial diagnostic hypotheses, reflecting their diagnostic breadth and the reliability of the generated candidate list. RFA was defined as the accuracy of the final decision produced after the model integrated supplementary clinical information. By analyzing the transition from initial differential accuracy (RTopD and RTop3D) to RFA, we assessed the models&#x2019; clinical reasoning and their capacity to refine diagnostic conclusions as the case progressed. Detailed definitions, assignment standards, and calculation formulas for these metrics are provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>The mean reciprocal rank (MRR) was used to evaluate the precision of the LLMs in ranking the correct diagnosis. The reciprocal rank for each case was determined by the position of the gold-standard diagnosis among the top 3 differential diagnoses provided by the model. The MRR was calculated using the following formula:</p><disp-formula id="equWL1"><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>M</mml:mi><mml:mi>R</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>C</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:mn>1</mml:mn><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where C corresponds to the number of cases on which the metric is evaluated, and <italic>r</italic><sub><italic>i</italic></sub> is the rank of the first occurrence of a correct answer in the final list for case <italic>i</italic>. Specifically, a score of 1, .5, or .33 was assigned if the correct diagnosis appeared as the first, second, or third suggestion, respectively. In any case where the correct diagnosis is ranked beyond the top 3 (<italic>r</italic><sub><italic>i</italic></sub>&#x003E;3) or is absent from the list, the contribution to the MRR is set to 0. This metric effectively rewards models that consistently place the gold-standard diagnosis at higher positions within the differential diagnosis list.</p></sec><sec id="s2-8"><title>Qualitative Error Analysis</title><p>To characterize recurrent patterns of diagnostic errors generated by the LLMs, a qualitative analysis was conducted on all cases in which the final diagnosis was incorrect. Following an inductive thematic coding process, 2 independent raters (JYW and XNL) reviewed the reasoning trajectories and iteratively identified five recurrent error domains: (1) failure to recognize specific features, involving omission of critical diagnostic clues or pathognomonic findings explicitly provided in the case description; (2) incorrect attribution, where clinical findings were identified but misinterpreted or assigned to an incorrect etiology; (3) failure in multisystem information integration, characterized by the inability to synthesize manifestations across multiple organ systems into a coherent systemic diagnosis; (4) frequency-based matching bias, referring to the tendency to prioritize high-prevalence diseases despite recognizing disease-specific clues suggestive of a rare condition; and (5) neglect of negative evidence, defined as the failure to incorporate exclusionary findings (eg, negative laboratory results) into differential diagnostic reasoning.</p><p>A multilabel coding approach was adopted, allowing a single diagnostic error to be assigned to multiple domains when applicable. Discrepancies between the primary raters were resolved by a third senior evaluator (JY). Given the nonmutually exclusive nature of these error categories, interrater reliability was assessed by decomposing the coding task into 5 independent binary classification problems. The category-specific Cohen kappa coefficients ranged from .824 to .939, with an average value of .898, indicating strong agreement in qualitative categorization.</p></sec><sec id="s2-9"><title>Quality Control</title><p>Several measures were adopted to enhance assessment reliability. Before formal evaluation, the 2 primary raters (JYW and XNL) conducted a calibration session using a representative subset of cases to align interpretations of the scoring criteria and the 5 predefined error domains. All assessments were performed in independent dialogue sessions using previously unused accounts to minimize potential information carryover between cases, and the raters were blinded to the identity of the LLMs generating the outputs, which were anonymized prior to evaluation. Residual disagreements were resolved through blinded adjudication by a third senior evaluator (JY). In addition, screenshots of all model outputs were archived to ensure traceability and allow subsequent verification of the extracted data.</p></sec><sec id="s2-10"><title>Statistical Analysis</title><p>Data were entered and organized in Excel, and statistical analyses were conducted using R software (version 4.6.0; R Foundation for Statistical Computing). Diagnostic accuracy was summarized as counts and percentages. There was no missing data in this study, as all 56 clinical vignettes were complete and all 4 LLMs successfully generated responses for each assigned case. Furthermore, no indeterminate or uninterpretable model outputs were encountered; all LLM responses were determinate and evaluable according to the predefined scoring criteria.</p><p>To account for within-case correlation, where each clinical vignette was evaluated by 4 LLMs across 2 diagnostic stages, yielding 448 individual diagnostic responses (the analytic unit), generalized estimating equations (GEE) with a binary logistic link and an exchangeable working correlation structure were fitted using the <italic>geepack package</italic> in R. An exchangeable structure was chosen because repeated observations within each case lacked a natural temporal order.</p><p>The GEE analyses proceeded in 2 stages. First, separate models were constructed for each primary endpoint (RTopD, RTop3D, and RFA). These models included LLM type and disease group (COPD vs RP) as main effects, along with their interaction (LLM &#x00D7; disease), to evaluate performance differences across disease categories. Second, to examine diagnostic refinement, a combined GEE model was fitted by pooling initial and final assessments and adding diagnostic stage as a within-subject factor. This model included the relevant main effects and key interactions (stage &#x00D7; LLM and stage &#x00D7; disease) to quantify changes in diagnostic accuracy. For all GEE models, results are reported as odds ratios (ORs) with 95% CIs, and pairwise comparisons of estimated marginal means were performed using Bonferroni adjustment. To assess the robustness of the primary statistical inferences, sensitivity analyses were conducted by comparing the primary exchangeable working correlation structure with an alternative independent working correlation structure. Consistency of parameter estimates, odds ratios, CIs, and statistical inferences across correlation specifications was examined to evaluate the robustness of the findings to the choice of working correlation structure.</p><p>To evaluate diagnostic ranking performance, the MRR was computed for each model as a descriptive summary metric. Given the repeated-measures design, in which the same clinical cases were evaluated across the 4 LLMs, and the nonnormal distribution of the ranking data, statistical comparisons were conducted using the Friedman test based on case-level reciprocal rank scores. When a statistically significant overall difference was detected, post-hoc pairwise comparisons were conducted using the Wilcoxon signed-rank test. To control the family-wise error rate across the 6 possible LLM pairs, the Bonferroni correction was applied (adjusted significance threshold: .008).</p><p>A 2-sided <italic>P</italic>&#x003C;.05 was considered statistically significant.</p></sec><sec id="s2-11"><title>Ethical Considerations</title><p>This study was conducted in accordance with the protocol approved by the Ethics Committee of Capital Medical University (approval number: 2025SY-166) and adhered to the ethical principles of the Declaration of Helsinki. The research used standardized clinical vignettes derived from a retrospective database for the purpose of benchmarking AI models. In alignment with institutional ethical guidelines for the use of deidentified, retrospective data, all patient information was fully anonymized prior to analysis to ensure that no individuals could be identified. No direct intervention or interaction with human participants occurred during this study. All research procedures were performed strictly following the data privacy and confidentiality standards approved by the institutional review board.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Baseline Characteristics of Clinical Cases</title><p>A total of 56 clinical vignettes were evaluated in this study, comprising 28 COPD cases and 28 RP cases. Across all cases, the mean age of the patients was 57.6 (SD 18.7) years, with a sex distribution of 43 males (76.8%) and 13 females (23.2%). In the COPD cohort, the mean age was 69.1 (SD 11.3) years (24 males and 4 females), while in the RP cohort, the mean age was 46.2 (SD 17.8) years (19 males and 9 females).</p></sec><sec id="s3-2"><title>Diagnostic Accuracy by LLMs</title><p>Across the 3 diagnostic performance metrics, differences in diagnostic accuracy were observed among the LLMs. Compared with Leftdoctor GPT, both DeepSeek and Doubao consistently showed higher diagnostic accuracy across all 3 metrics (RTop3D: OR=4.50, 95% CI 1.71-11.83, <italic>P</italic>=.002; RTopD: OR=6.16 and OR=7.22, 95% CI 1.62-23.36 and 1.86-28.03, <italic>P</italic>=.008 and <italic>P</italic>=.004 for DeepSeek and Doubao, respectively; final diagnosis: OR=6.25 and 9.17, 95% CI 2.55-15.34 and 3.38-24.90 for DeepSeek and Doubao, respectively, all <italic>P</italic>&#x003C;.001). In contrast, KIMI did not differ significantly from Leftdoctor GPT for any of the metrics (all <italic>P</italic>&#x003E;.05). Overall, DeepSeek and Doubao demonstrated better diagnostic performance, whereas the performance of KIMI was comparable to that of Leftdoctor GPT (<xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Generalized estimating equations analysis of diagnostic accuracy by large language model, disease, and diagnostic stage. Odds ratio with 95% CIs were estimated using generalized estimating equations with an exchangeable working correlation structure to account for repeated measurements within the same cases. The model included main effects of large language model, disease group, and diagnostic stage, as well as interaction terms (large language model &#x00D7; disease, disease &#x00D7; stage, and large language model &#x00D7; stage). Reference categories are indicated in the &#x201C;Reference&#x201D; column. Odds ratios greater than 1 indicate higher odds of correct diagnosis compared with the reference group. All <italic>P</italic> values are two-sided.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variable, metric, and comparison</td><td align="left" valign="bottom">Reference</td><td align="left" valign="bottom">OR<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> values</td></tr></thead><tbody><tr><td align="left" valign="top">LLM<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">Leftdoctor GPT</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTop3D<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek</td><td align="left" valign="top"/><td align="left" valign="top">4.50 (1.71-11.83)</td><td align="left" valign="top">.002</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Doubao</td><td align="left" valign="top"/><td align="left" valign="top">4.50 (1.71-11.83)</td><td align="left" valign="top">.002</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>KIMI</td><td align="left" valign="top"/><td align="left" valign="top">1.30 (0.53-3.21)</td><td align="left" valign="top">.56</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTopD<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek</td><td align="left" valign="top"/><td align="left" valign="top">6.16 (1.62-23.36)</td><td align="left" valign="top">.008</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Doubao</td><td align="left" valign="top"/><td align="left" valign="top">7.22 (1.86-28.03)</td><td align="left" valign="top">.004</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>KIMI</td><td align="left" valign="top"/><td align="left" valign="top">1.56 (0.66-3.70)</td><td align="left" valign="top">.31</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFA<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek</td><td align="left" valign="top"/><td align="left" valign="top">6.25 (2.55-15.34)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Doubao</td><td align="left" valign="top"/><td align="left" valign="top">9.17 (3.38-24.90)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>KIMI</td><td align="left" valign="top"/><td align="left" valign="top">0.83 (0.45-1.54)</td><td align="left" valign="top">.56</td></tr><tr><td align="left" valign="top">Disease</td><td align="left" valign="top">RP</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTop3D</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>COPD<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top">50 (10.11-247.23)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTopD</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>COPD</td><td align="left" valign="top"/><td align="left" valign="top">108.33 (16.67-703.98)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFA<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>COPD</td><td align="left" valign="top"/><td align="left" valign="top">11.50 (3.24-40.86)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Stage</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No metric</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Top-1 Differential Diagnosis Stage</td><td align="left" valign="top">Final Assessment Stage</td><td align="left" valign="top">0.307 (0.09-1.02)</td><td align="left" valign="top">.05</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No metric</td><td align="left" valign="top">Leftdoctor GPT &#x00D7; Final Assessment Stage</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek &#x00D7; Top-1 Differential Diagnosis Stage</td><td align="left" valign="top"/><td align="left" valign="top">0.60 (0.20-1.85)</td><td align="left" valign="top">.38</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Doubao&#x00D7; Top-1 Differential Diagnosis Stage</td><td align="left" valign="top"/><td align="left" valign="top">0.49 (0.18-1.34)</td><td align="left" valign="top">.16</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>KIMI &#x00D7; Top-1 Differential Diagnosis Stage</td><td align="left" valign="top"/><td align="left" valign="top">0.87 (0.35-2.16)</td><td align="left" valign="top">.77</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No metric</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>COPD &#x00D7; Top-1 Differential Diagnosis Stage</td><td align="left" valign="top">RP&#x00D7; Final Assessment Stage</td><td align="left" valign="top">4.10 (0.88-19.11)</td><td align="left" valign="top">.07</td></tr><tr><td align="left" valign="top">Interaction terms</td><td align="left" valign="top">Leftdoctor GPT &#x00D7; RP</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTop3D</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Doubao &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">0.12 (0.03-0.56)</td><td align="left" valign="top">.006</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">0.22 (0.04-1.28)</td><td align="left" valign="top">.09</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>KIMI &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">0.77 (0.34-4.16)</td><td align="left" valign="top">.70</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTopD</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Doubao &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">0.08 (0.01-0.46)</td><td align="left" valign="top">.005</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">0.16 (0.02-1.18)</td><td align="left" valign="top">.07</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>KIMI &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">0.64 (0.17-2.47)</td><td align="left" valign="top">.52</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFA</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Doubao &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">0.20 (0.06-0.71)</td><td align="left" valign="top">.01</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">0.45 (0.08-2.68)</td><td align="left" valign="top">.38</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>KIMI &#x00D7; COPD</td><td align="left" valign="top"/><td align="left" valign="top">3.39 (0.91-12.63)</td><td align="left" valign="top">.07</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>OR: odds ratio.</p></fn><fn id="table1fn2"><p><sup>b</sup>LLM: large language model.</p></fn><fn id="table1fn3"><p><sup>c</sup>RTop3D: proportion of cases in which the correct diagnosis was ranked within the top three differential diagnoses.</p></fn><fn id="table1fn4"><p><sup>d</sup>RTopD: proportion of cases in which the correct diagnosis was ranked first.</p></fn><fn id="table1fn5"><p><sup>e</sup>RFA: proportion of cases in which the final diagnosis was correctly identified.</p></fn><fn id="table1fn6"><p><sup>f</sup>COPD: chronic obstructive pulmonary disease.</p></fn><fn id="table1fn7"><p><sup>g</sup>RP: relapsing polychondritis.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Diagnostic Accuracy by Disease</title><p>Significant differences in diagnostic accuracy were observed between the 2 diseases across all 3 metrics. Compared with RP, diagnostic accuracy was substantially higher for COPD for RTop3D (OR=50; 95% CI 10.11 -247.23; <italic>P</italic>&#x003C;.001), RTopD (OR=108.33; 95% CI 16.67 -703.98; <italic>P</italic>&#x003C;.001), and RFA (OR=11.50; 95% CI 3.24 -40.86<italic>; P</italic>&#x003C;.001). These findings indicate that the LLMs achieved markedly better diagnostic performance for COPD cases than for RP cases (<xref ref-type="table" rid="table1">Table 1</xref>).</p></sec><sec id="s3-4"><title>Changes in Diagnostic Accuracy With Additional Clinical Information</title><p>The incorporation of additional clinical information led to an overall upward trend in diagnostic accuracy, although the difference between the top-1 differential diagnosis and final assessment stages was of borderline statistical significance (<italic>P</italic>=.05) (<xref ref-type="table" rid="table1">Table 1</xref>). Furthermore, no significant interaction was observed between diagnostic stage and LLM type (all <italic>P</italic>&#x003E;.05), indicating that the models responded similarly to the supplementary data.</p><p>Despite the lack of a strictly significant interaction between diagnostic stage and disease type (<italic>P</italic>=.07), descriptive results revealed distinct performance trajectories between the 2 disease categories (<xref ref-type="table" rid="table2">Table 2</xref>). For COPD, all LLMs already achieved high diagnostic accuracy early in the top-1 differential diagnosis stage (82%&#x2010;93%), with minimal further improvement after additional clinical information was supplied. In contrast, diagnostic accuracy for RP was substantially lower at the top-1 differential diagnosis stage (7%&#x2010;36%) but improved markedly by the final assessment stage (25%&#x2010;79%). These findings suggest that the overall improvement across diagnostic stages was primarily driven by marked gains in RP accuracy, whereas diagnostic performance for COPD remained consistently high throughout the diagnostic process.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Accuracy comparison of LLMs<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> in common and rare diseases. Values are presented as n (%), with percentages calculated out of 28 cases per disease.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">DeepSeek, n (%)</td><td align="left" valign="bottom">KIMI, n (%)</td><td align="left" valign="bottom">Doubao, n (%)</td><td align="left" valign="bottom">Leftdoctor GPT, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">COPD<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTop3D<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">25 (89.29)</td><td align="left" valign="top">25 (89.29)</td><td align="left" valign="top">23 (82.14)</td><td align="left" valign="top">25 (89.29)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTopD<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">25 (89.29)</td><td align="left" valign="top">25 (89.29)</td><td align="left" valign="top">23 (82.14)</td><td align="left" valign="top">25 (89.29)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFA<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">26 (92.86)</td><td align="left" valign="top">26 (92.86)</td><td align="left" valign="top">25 (89.29)</td><td align="left" valign="top">23 (82.14)</td></tr><tr><td align="left" valign="top" colspan="5">RP<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTop3D</td><td align="left" valign="top">12 (42.86)</td><td align="left" valign="top">5 (17.86)</td><td align="left" valign="top">12 (42.86)</td><td align="left" valign="top">4 (14.29)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RTopD</td><td align="left" valign="top">9 (32.14)</td><td align="left" valign="top">3 (10.71)</td><td align="left" valign="top">10 (35.71)</td><td align="left" valign="top">2 (7.14)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFA</td><td align="left" valign="top">20 (71.43)</td><td align="left" valign="top">7 (25.00)</td><td align="left" valign="top">22 (78.57)</td><td align="left" valign="top">8 (28.57)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>LLM: large language model. </p></fn><fn id="table2fn2"><p><sup>b</sup>COPD: chronic obstructive pulmonary disease. </p></fn><fn id="table2fn3"><p><sup>c</sup>RTop3D=proportion of cases in which the correct diagnosis was ranked within the top three differential diagnoses. </p></fn><fn id="table2fn4"><p><sup>d</sup>RTopD=proportion of cases in which the correct diagnosis was ranked first. </p></fn><fn id="table2fn5"><p><sup>e</sup>RFA=proportion of cases in which the final diagnosis was correctly identified. </p></fn><fn id="table2fn6"><p><sup>f</sup>RP: relapsing polychondritis.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Interaction Between Disease Type and Model</title><p>A significant interaction between disease type and LLM was observed for all diagnostic metrics, including RTop3D (Wald &#x03C7;&#x00B2;<sub>3</sub>=9.42; <italic>P</italic>=.02), RTopD (Wald &#x03C7;&#x00B2;<sub>3</sub>=8.52; <italic>P</italic>=.04), and RFA (Wald &#x03C7;&#x00B2;<sub>3</sub>=18.62; <italic>P</italic>&#x003C;.001), indicating that differences in LLM performance varied across disease.</p><p>Parameter estimates from the GEE model indicated that this interaction was mainly associated with Doubao. Specifically, the interaction between Doubao and COPD was statistically significant across all three diagnostic metrics: the RTop3D metric (OR=.12, 95% CI .03-.56; <italic>P</italic>=.006) and the RTopD metric (OR=.08, 95% CI .01-.46; <italic>P</italic>=.005), as well as for the RFA metric (OR=.20, 95% CI .06-.71; <italic>P</italic>=.01). In contrast, the interaction terms for DeepSeek and KIMI were not statistically significant (all <italic>P</italic>&#x003E;.05).</p><p>These findings suggest that the variation in diagnostic performance across diseases was mainly attributable to disease-specific performance differences in the Doubao model.</p></sec><sec id="s3-6"><title>LLM Differential Diagnosis Ranking Performance (MRR)</title><p>Ranking performance, which accounts for both the correctness and rank position of differential diagnoses, was evaluated using MRR.</p><p>For COPD diagnosis, DeepSeek, KIMI, and Leftdoctor GPT achieved the highest MRR scores (.89 (SD .32)), followed by Doubao (mean .82, SD .39). Given the repeated-measures design of the evaluation across the same clinical cases, the Friedman test was used. The analysis indicated no statistically significant difference in diagnostic ranking performance among the 4 LLMs (&#x03C7;&#x00B2;<sub>3</sub>=1.385; <italic>P</italic>=.71; Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>For RP diagnosis, diagnostic ranking performance differed notably across the LLMs. Doubao achieved the highest MRR score (.39, SD .48), followed by DeepSeek (.37, SD .46), KIMI (.14, SD .32), and Leftdoctor GPT (.10, SD .28). Given the repeated-measures design, a Friedman test was conducted based on case-level performance scores, revealing a statistically significant overall difference among the models (&#x03C7;&#x00B2;<sub>3</sub>=21.933; <italic>P</italic>&#x003C;.001). Post-hoc pairwise comparisons using the Wilcoxon signed-rank test, also performed at the case level, with Bonferroni correction (adjusted significance threshold of <italic>&#x03B1;</italic>=.0083 for 6 comparisons), indicated that both Doubao and DeepSeek significantly outperformed Leftdoctor GPT (both unadjusted <italic>P</italic>=.002). Furthermore, Doubao significantly outperformed KIMI (unadjusted <italic>P</italic>=.008). Although DeepSeek scored higher than KIMI, this difference did not reach the adjusted significance threshold after Bonferroni correction (unadjusted <italic>P</italic>=.01). Differences between all other LLM pairs were not statistically significant (<xref ref-type="fig" rid="figure3">Figure 3</xref>, Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Mean reciprocal rank across large language models for chronic obstructive pulmonary disease and relapsing polychondritis. Bars represent mean and standard error of the mean (n=28 cases per group). Global comparisons among the four models were performed using the Friedman test based on case-level performance scores. Post hoc pairwise comparisons were conducted via the Wilcoxon signed-rank test, also performed at the case level, with Bonferroni correction for multiple testing. Asterisks (*) indicate statistically significant differences (<italic>P</italic>&#x003C;.05); ns indicates no significant difference (<italic>P</italic>&#x003E;.05). SEM: standard error of the mean; MRR: mean reciprocal rank; LLMs: large language models; COPD: chronic obstructive pulmonary disease; RP: relapsing polychondritis.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e89963_fig03.png"/></fig></sec><sec id="s3-7"><title>Qualitative Analysis of Diagnostic Errors</title><p>Distinct error patterns were observed between COPD and RP cases (<xref ref-type="table" rid="table3">Table 3</xref>). Among the 12 incorrect COPD diagnoses, errors were limited to failure to recognize specific features (10, 83.3%) and incorrect attribution (2, 16.7%), with no concurrent error types identified.</p><p>Among the 55 incorrect RP diagnoses, a total of 61 error events were identified, with 6 cases (10.9%) involving multiple error types. The most common errors were neglect of negative evidence (21, 34.4%) and failure to recognize specific features (14, 23%), followed by incorrect attribution (9, 14.8%), frequency-based matching bias (9, 14.8%), and failure in multisystem information integration (8, 13.1%).</p><p>Model-specific differences were also observed. Leftdoctor GPT showed the highest number of errors (21), most commonly neglect of negative evidence (10). Kimi demonstrated the broadest distribution of error categories (21), whereas DeepSeek errors were mainly related to neglect of negative evidence and failure to recognize specific features (5 each). Doubao showed the fewest errors (6), without a dominant error category.</p><p>Sensitivity analyses were conducted using alternative independent working correlation structures. The resulting parameter estimates, ORs, CIs, and statistical inferences were materially consistent with those obtained under the primary exchangeable structure, indicating that the study findings were robust to the choice of working correlation structure (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Distribution of diagnostic error domains across four LLMs for COPD and RP cases.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Error domain</td><td align="left" valign="bottom">DeepSeek</td><td align="left" valign="bottom">Kimi</td><td align="left" valign="bottom">Doubao</td><td align="left" valign="bottom">Leftdoctor GPT</td><td align="left" valign="bottom">Total, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">COPD<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> (n=12 error events)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Failure to recognize specific features</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">4</td><td align="left" valign="top">10 (83.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Incorrect attribution</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td><td align="left" valign="top">1</td><td align="left" valign="top">1</td><td align="left" valign="top">2 (16.7)</td></tr><tr><td align="left" valign="top" colspan="6">RP<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> (n=61 error events)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neglect of negative evidence</td><td align="left" valign="top">5</td><td align="left" valign="top">5</td><td align="left" valign="top">1</td><td align="left" valign="top">10</td><td align="left" valign="top">21 (34.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Failure to recognize specific features</td><td align="left" valign="top">5</td><td align="left" valign="top">5</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">14 (23)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Incorrect attribution</td><td align="left" valign="top">1</td><td align="left" valign="top">4</td><td align="left" valign="top">1</td><td align="left" valign="top">3</td><td align="left" valign="top">9 (14.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Frequency-based matching bias</td><td align="left" valign="top">0</td><td align="left" valign="top">4</td><td align="left" valign="top">1</td><td align="left" valign="top">4</td><td align="left" valign="top">9 (14.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Failure in multisystem information integration</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td><td align="left" valign="top">1</td><td align="left" valign="top">2</td><td align="left" valign="top">8 (13.1)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>COPD: chronic obstructive pulmonary disease.</p></fn><fn id="table3fn2"><p><sup>b</sup>RP: relapsing polychondritis.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study evaluated the diagnostic performance of 4 Chinese LLMs across common and rare diseases using a stepwise, hypothetico-deductive framework. Three main findings emerged. First, all models demonstrated substantially higher diagnostic accuracy for the common disease (COPD) compared with the rare disease (RP). Second, incremental clinical information improved diagnostic accuracy primarily in rare disease scenarios, with considerable variation across models. Third, differences between models were minimal under relatively straightforward diagnostic conditions but became more pronounced in diagnostically challenging cases.</p></sec><sec id="s4-2"><title>Diagnostic Performance Disparity Between Common and Rare Diseases</title><p>A prominent finding of this study is the consistent performance gap between common and rare diseases across all evaluated models. Similar patterns have been reported in recent studies, where LLM performance declines in complex or rare disease scenarios [<xref ref-type="bibr" rid="ref32">32</xref>]. While this phenomenon is often attributed to differences in data availability [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>], our results suggest that it may also relate to the distinct clinical structures of these conditions and the influence of learned associations on model outputs.</p><p>For common diseases such as COPD, clinical presentations tend to align with frequently encountered and well-represented patterns in training data. Qualitative observations show that errors in the COPD group were largely confined to the failure to identify specific features, while instances of incorrect attribution were infrequent across multiple evaluations. This suggests that when key diagnostic information is correctly recognized, models can generally map it to the intended diagnosis. In these situations, LLMs can rely on strong probabilistic associations to produce accurate and stable diagnostic suggestions [<xref ref-type="bibr" rid="ref35">35</xref>].</p><p>The disparity between COPD and RP extends to differences in organ system involvement and diagnostic breadth. While COPD is primarily localized to the respiratory system, RP is a systemic condition characterized by multiorgan involvement and a complex symptom profile [<xref ref-type="bibr" rid="ref25">25</xref>]. This complexity requires the synthesis of heterogeneous signals, a task where LLMs showed limitations. Our analysis identified a higher diversity of error types in RP cases, including the neglect of negative evidence and failures in multisystem integration. The presence of frequency-based matching bias, where models prioritized high-prevalence conditions despite identifying RP-specific cues, indicates that a reliance on learned associations may affect the rigorous refinement required for complex, multisystemic conditions [<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>Consequently, their performance remains less stable in low-prevalence or diagnostically complex scenarios, which has important implications for their safe clinical application [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>].</p></sec><sec id="s4-3"><title>The Role of Incremental Information in Diagnostic Refinement</title><p>This study also provides insight into how LLMs respond to sequentially provided clinical information. While additional data had a limited impact on diagnostic accuracy for COPD, it significantly improved performance in RP cases for some models. This asymmetry suggests that the use of incremental information depends on the level of initial diagnostic certainty [<xref ref-type="bibr" rid="ref39">39</xref>]. In the COPD group, the high initial diagnostic accuracy across models left limited room for further improvement through incremental data. Although infrequent diagnostic errors occurred in the final assessment stage, such as the omission of specific features in either the history or laboratory results, the strong initial signals associated with common conditions allowed LLMs to maintain stable performance, largely unaffected by the addition of new information.</p><p>In contrast, when initial diagnostic signals are weak or ambiguous, as in rare diseases, the introduction of new information can meaningfully shift the probability distribution and improve diagnostic accuracy [<xref ref-type="bibr" rid="ref40">40</xref>]. However, the qualitative observations indicate that while incremental information facilitates performance gains, it does not fully resolve the reasoning challenges inherent in complex diagnoses. Even with access to comprehensive laboratory and imaging results, LLMs still exhibit systematic flaws in the final reasoning stage, most notably the neglect of negative evidence [<xref ref-type="bibr" rid="ref36">36</xref>]. This suggests that diagnostic failures in rare disease scenarios are linked not only to information scarcity but also to difficulties in the logical processing of exclusionary findings. Therefore, simply increasing the volume of information input may be insufficient to overcome the underlying reasoning patterns that limit LLM performance in complex diagnostic tasks.</p></sec><sec id="s4-4"><title>Model-Specific Differences Under Diagnostic Complexity</title><p>Another important observation is that differences between models were relatively small in COPD but became more pronounced in RP. This suggests that model performance may converge under conditions where diagnostic patterns are clear and well-represented, but diverge when cases require handling uncertainty or integrating less typical information [<xref ref-type="bibr" rid="ref41">41</xref>].</p><p>In relatively straightforward scenarios, most models are able to generate similar high-probability outputs, resulting in comparable performance. However, in more complex or ambiguous cases, models may differ in how they prioritize competing diagnostic possibilities or respond to incomplete information. These differences may reflect variations in training data composition, model architecture, or alignment strategies [<xref ref-type="bibr" rid="ref42">42</xref>].</p><p>Qualitative observations provide preliminary insights into these model-specific patterns. In RP diagnosis, Leftdoctor GPT exhibited a higher frequency of errors, particularly in the neglect of negative evidence, which may indicate challenges in integrating exclusionary information. Kimi showed a broader distribution across error categories, whereas DeepSeek&#x2019;s errors were more concentrated in feature identification and the neglect of negative evidence. In contrast, Doubao achieved the highest diagnostic accuracy in the RP group with the fewest total errors and no dominant error type, suggesting a relatively balanced performance across different reasoning domains. It is important to note that, given the limited number of error events in this qualitative analysis, these observations should be regarded as preliminary rather than definitive conclusions about model capabilities. Further validation in larger-scale studies is required to confirm these behavioral patterns.</p></sec><sec id="s4-5"><title>Clinical and Research Implications</title><p>The findings of this study suggest that the clinical use of LLMs may be context-dependent. While these models provide relatively reliable diagnostic support for common diseases with typical presentations, their outputs in rare or atypical cases require careful interpretation due to lower stability and higher intermodel variability.</p><p>From a clinical perspective, the identified error patterns offer practical guidance for diagnostic assistance. Clinicians should specifically verify whether AI-generated suggestions have appropriately integrated all exclusionary findings to ensure that sufficient diagnostic breadth is maintained for complex presentations. From a research perspective, these results underscore the importance of evaluation frameworks that reflect real-world, stepwise clinical reasoning. Such approaches may provide more realistic estimates of model performance compared to single-step evaluations based on complete information. Additionally, the preliminary error profiles identified in this study can guide targeted improvements in diagnostic capabilities. Future work could explore specialized prompting strategies or fine-tuning methods to help models weigh negative evidence and integrate multisystem clinical information more effectively. Beyond diagnostic reasoning, advancing Chinese LLMs will require expanding their applications across diverse medical specialties and complex decision-making scenarios. Integrating multimodal data such as medical imaging, pathology, and genetic information may help address the instability and information dependence observed in rare disease reasoning [<xref ref-type="bibr" rid="ref43">43</xref>]. Specialty-specific optimization and cross-disease validation are critical for enhancing generalizability and reliability, while assessment frameworks that incorporate error patterns can provide a more comprehensive measure of clinical reasoning than accuracy metrics alone. Because demographic information such as age and sex was preserved in the clinical vignettes, the observed diagnostic performance may partially reflect how LLMs use demographic cues during diagnostic reasoning. Future studies should further investigate the extent to which demographic characteristics influence diagnostic outputs and whether demographic-related biases exist across different models.</p></sec><sec id="s4-6"><title>Limitations</title><p>This study has several limitations. First, the sample size was relatively small, leading to sparse-data instability in some statistical estimates, as reflected by large ORs with wide CIs. Although the sample size is consistent with prior exploratory studies of LLM diagnostic performance, larger multicenter studies are warranted to validate these findings across more diverse clinical scenarios [<xref ref-type="bibr" rid="ref32">32</xref>]. Second, all models were tested using a standardized single prompt, which ensured fair comparisons but may not reflect their performance under alternative prompting strategies. In addition, each prompt was evaluated in a single run, which does not allow assessment of stochastic variability in model outputs; multiple runs could have better quantified or mitigated random fluctuations inherent to LLM-based systems. Third, our clinical vignettes were curated to exclude major comorbidities and included only confirmed cases with complete information. While we naturalized the language to reflect real patient narratives rather than textbook-style descriptions, this design may yield performance estimates that are higher than those observed in unselected real-world clinical practice. Fourth, while the vignettes were paraphrased, the use of public datasets for benchmarking involves an inherent, unquantified risk of semantic contamination. Since language models operate on semantic embeddings rather than relying solely on exact string matches, text modification may not entirely preclude the recognition of cases encountered during training. Finally, our model selection represents a snapshot based on benchmark rankings during the study period and does not encompass all representative Chinese LLMs. Given the rapid evolution of AI technology, performance rankings may shift as new models are released, and future research should incorporate a broader, more up-to-date range of models to provide a more comprehensive evaluation of their evolving capabilities.</p></sec><sec id="s4-7"><title>Conclusions</title><p>This study evaluated the diagnostic capabilities of 4 Chinese LLMs for common (COPD) and rare (RP) diseases using a step-by-step, clinical vignette-based approach. LLMs demonstrated strong diagnostic performance for common diseases but substantially lower and more variable performance for rare diseases. This performance disparity reflects the interplay between training data prevalence and the inherent challenges of synthesizing multisystemic clinical information. Although stepwise clinical information improved accuracy in rare disease cases, it did not eliminate the underlying reasoning limitations, particularly in integrating information and weighing evidence, which remained even when complete data were available. These findings suggest that current LLMs may perform well in pattern-consistent conditions but remain limited in handling diagnostic uncertainty and complexity. Careful evaluation and appropriate clinical oversight are therefore essential for their safe application in practice.</p></sec></sec></body><back><ack><p>We would like to thank all the participants for their support in developing this paper.</p></ack><notes><sec><title>Funding</title><p>This research is supported by the National Natural Science Foundation of China (72574152).</p></sec><sec><title>Data Availability</title><p>All data generated or analyzed during this study are with the corresponding author. She is available to answer any questions about the datasets.</p></sec></notes><fn-group><fn fn-type="con"><p>JYW, JY, and RG designed and conducted the research. JYW completed the data acquisition, data analysis, and wrote the first draft of the manuscript. JYW, JY, and RG were responsible for supervising the data analysis and manuscript writing. All authors contributed to the revision of the article and approved the final draft submitted.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CMCR</term><def><p>China Clinical Case Results Database</p></def></def-item><def-item><term id="abb2">COPD</term><def><p>chronic obstructive pulmonary disease</p></def></def-item><def-item><term id="abb3">DDx</term><def><p>differential diagnosis</p></def></def-item><def-item><term id="abb4">GEE</term><def><p>generalized estimating equation</p></def></def-item><def-item><term id="abb5">GOLD</term><def><p>Global Initiative for Chronic Obstructive Lung Disease</p></def></def-item><def-item><term id="abb6"><italic>ICD-10</italic></term><def><p><italic>International Classification of Diseases, Tenth Revision</italic></p></def></def-item><def-item><term id="abb7">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb8">MCQ</term><def><p>multiple choice question</p></def></def-item><def-item><term id="abb9">MRR</term><def><p>mean reciprocal rank</p></def></def-item><def-item><term id="abb10">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb11">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb12">RFA</term><def><p>rate of final diagnostic accuracy</p></def></def-item><def-item><term id="abb13">RP</term><def><p>relapsing polychondritis</p></def></def-item><def-item><term id="abb14">RTop3D</term><def><p>rate of final diagnosis within the top 3 DDx list</p></def></def-item><def-item><term id="abb15">RTopD</term><def><p>rate of final diagnosis as top diagnosis</p></def></def-item><def-item><term id="abb16">STARD</term><def><p>Standards for Reporting Diagnostic Accuracy Studies</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Qi</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Artificial intelligence revolutionizes anti&#x2010;infective drug discovery: from target identification to lead optimization</article-title><source>iMetaMed</source><year>2025</year><month>12</month><volume>1</volume><issue>2</issue><fpage>e70011</fpage><pub-id pub-id-type="doi">10.1002/imm3.70011</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Revolutionizing health care: the transformative impact of large language models in medicine</article-title><source>J Med Internet Res</source><year>2025</year><month>01</month><day>7</day><volume>27</volume><fpage>e59069</fpage><pub-id pub-id-type="doi">10.2196/59069</pub-id><pub-id pub-id-type="medline">39773666</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>M</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Park</surname><given-names>YJ</given-names> </name><name name-style="western"><surname>Paget</surname><given-names>M</given-names> </name><name name-style="western"><surname>Naugler</surname><given-names>C</given-names> </name></person-group><article-title>Automated paper screening for clinical reviews using large language models: data analysis study</article-title><source>J Med Internet Res</source><year>2024</year><month>01</month><day>12</day><volume>26</volume><fpage>e48996</fpage><pub-id pub-id-type="doi">10.2196/48996</pub-id><pub-id pub-id-type="medline">38214966</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>Q</given-names> </name></person-group><article-title>Physician versus large language model chatbot responses to web-based questions from autistic patients in Chinese: cross-sectional comparative analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>04</month><day>30</day><volume>26</volume><fpage>e54706</fpage><pub-id pub-id-type="doi">10.2196/54706</pub-id><pub-id pub-id-type="medline">38687566</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name></person-group><article-title>Utility of ChatGPT in clinical practice</article-title><source>J Med Internet Res</source><year>2023</year><month>06</month><day>28</day><volume>25</volume><fpage>e48568</fpage><pub-id pub-id-type="doi">10.2196/48568</pub-id><pub-id pub-id-type="medline">37379067</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chakraborty</surname><given-names>C</given-names> </name><name name-style="western"><surname>Pal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bhattacharya</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dash</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SS</given-names> </name></person-group><article-title>Overview of chatbots with special emphasis on artificial intelligence-enabled ChatGPT in medical science</article-title><source>Front Artif Intell</source><year>2023</year><volume>6</volume><fpage>1237704</fpage><pub-id pub-id-type="doi">10.3389/frai.2023.1237704</pub-id><pub-id pub-id-type="medline">38028668</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barak-Corren</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wolf</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rozenblum</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Harnessing the power of generative AI for clinical summaries: perspectives from emergency physicians</article-title><source>Ann Emerg Med</source><year>2024</year><month>08</month><volume>84</volume><issue>2</issue><fpage>128</fpage><lpage>138</lpage><pub-id pub-id-type="doi">10.1016/j.annemergmed.2024.01.039</pub-id><pub-id pub-id-type="medline">38483426</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blease</surname><given-names>C</given-names> </name><name name-style="western"><surname>Worthen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name></person-group><article-title>Psychiatrists&#x2019; experiences and opinions of generative artificial intelligence in mental healthcare: an online mixed methods survey</article-title><source>Psychiatry Res</source><year>2024</year><month>03</month><volume>333</volume><fpage>115724</fpage><pub-id pub-id-type="doi">10.1016/j.psychres.2024.115724</pub-id><pub-id pub-id-type="medline">38244285</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krusche</surname><given-names>M</given-names> </name><name name-style="western"><surname>Callhoff</surname><given-names>J</given-names> </name><name name-style="western"><surname>Knitza</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ruffer</surname><given-names>N</given-names> </name></person-group><article-title>Diagnostic accuracy of a large language model in rheumatology: comparison of physician and ChatGPT-4</article-title><source>Rheumatol Int</source><year>2024</year><month>02</month><volume>44</volume><issue>2</issue><fpage>303</fpage><lpage>306</lpage><pub-id pub-id-type="doi">10.1007/s00296-023-05464-6</pub-id><pub-id pub-id-type="medline">37742280</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Balogh</surname><given-names>EP</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>BT</given-names> </name><name name-style="western"><surname>Ball</surname><given-names>JR</given-names> </name></person-group><source>Improving Diagnosis in Health Care</source><year>2015</year><publisher-name>National Academies Press (US)</publisher-name><pub-id pub-id-type="doi">10.17226/21794</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="medline">37438534</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sandmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Riepenhausen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Plagwitz</surname><given-names>L</given-names> </name><name name-style="western"><surname>Varghese</surname><given-names>J</given-names> </name></person-group><article-title>Systematic analysis of ChatGPT, Google search and Llama 2 for clinical decision support tasks</article-title><source>Nat Commun</source><year>2024</year><month>03</month><day>6</day><volume>15</volume><issue>1</issue><fpage>2050</fpage><pub-id pub-id-type="doi">10.1038/s41467-024-46411-8</pub-id><pub-id pub-id-type="medline">38448475</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Nadeau</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kroutikov</surname><given-names>M</given-names> </name><name name-style="western"><surname>McNeil</surname><given-names>K</given-names> </name><name name-style="western"><surname>Baribeau</surname><given-names>S</given-names> </name></person-group><article-title>Benchmarking Llama2, mistral, gemma and GPT for factuality, toxicity</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 15, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2404.09785</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>G&#x00FC;nay</surname><given-names>S</given-names> </name><name name-style="western"><surname>&#x00D6;zt&#x00FC;rk</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yi&#x011F;it</surname><given-names>Y</given-names> </name></person-group><article-title>The accuracy of Gemini, GPT-4, and GPT-4o in ECG analysis: a comparison with cardiologists and emergency medicine specialists</article-title><source>Am J Emerg Med</source><year>2024</year><month>10</month><volume>84</volume><fpage>68</fpage><lpage>73</lpage><pub-id pub-id-type="doi">10.1016/j.ajem.2024.07.043</pub-id><pub-id pub-id-type="medline">39096711</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sonoda</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kurokawa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Nakamura</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Diagnostic performances of GPT-4o, Claude 3 Opus, and Gemini 1.5 Pro in &#x201C;Diagnosis Please&#x201D; cases</article-title><source>Jpn J Radiol</source><year>2024</year><month>11</month><volume>42</volume><issue>11</issue><fpage>1231</fpage><lpage>1235</lpage><pub-id pub-id-type="doi">10.1007/s11604-024-01619-y</pub-id><pub-id pub-id-type="medline">38954192</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ying</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Screening/diagnosis of pediatric endocrine disorders through the artificial intelligence model in different language settings</article-title><source>Eur J Pediatr</source><year>2024</year><month>06</month><volume>183</volume><issue>6</issue><fpage>2655</fpage><lpage>2661</lpage><pub-id pub-id-type="doi">10.1007/s00431-024-05527-1</pub-id><pub-id pub-id-type="medline">38502320</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ling</surname><given-names>W</given-names> </name></person-group><article-title>Performance of artificial intelligence chatbots on ultrasound examinations: cross-sectional comparative analysis</article-title><source>JMIR Med Inform</source><year>2025</year><month>01</month><day>9</day><volume>13</volume><fpage>e63924</fpage><pub-id pub-id-type="doi">10.2196/63924</pub-id><pub-id pub-id-type="medline">39814698</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aljindan</surname><given-names>FK</given-names> </name><name name-style="western"><surname>Al Qurashi</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Albalawi</surname><given-names>IAS</given-names> </name><etal/></person-group><article-title>ChatGPT conquers the Saudi medical licensing exam: exploring the accuracy of artificial intelligence in medical knowledge assessment and implications for modern medical education</article-title><source>Cureus</source><year>2023</year><month>09</month><volume>15</volume><issue>9</issue><fpage>e45043</fpage><pub-id pub-id-type="doi">10.7759/cureus.45043</pub-id><pub-id pub-id-type="medline">37829968</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alessandri Bonetti</surname><given-names>M</given-names> </name><name name-style="western"><surname>Giorgino</surname><given-names>R</given-names> </name><name name-style="western"><surname>Gallo Afflitto</surname><given-names>G</given-names> </name><name name-style="western"><surname>De Lorenzi</surname><given-names>F</given-names> </name><name name-style="western"><surname>Egro</surname><given-names>FM</given-names> </name></person-group><article-title>How does ChatGPT perform on the Italian residency admission national exam compared to 15,869 medical graduates?</article-title><source>Ann Biomed Eng</source><year>2024</year><month>04</month><volume>52</volume><issue>4</issue><fpage>745</fpage><lpage>749</lpage><pub-id pub-id-type="doi">10.1007/s10439-023-03318-7</pub-id><pub-id pub-id-type="medline">37490183</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ebrahimian</surname><given-names>M</given-names> </name><name name-style="western"><surname>Behnam</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ghayebi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Sobhrakhshankhah</surname><given-names>E</given-names> </name></person-group><article-title>ChatGPT in Iranian medical licensing examination: evaluating the diagnostic accuracy and decision-making capabilities of an AI-based model</article-title><source>BMJ Health Care Inform</source><year>2023</year><month>12</month><day>11</day><volume>30</volume><issue>1</issue><fpage>e100815</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2023-100815</pub-id><pub-id pub-id-type="medline">38081765</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>W</given-names> </name><etal/></person-group><article-title>How does ChatGPT-4 preform on non-English national medical licensing examination? An evaluation in Chinese language</article-title><source>PLOS Digit Health</source><year>2023</year><month>12</month><volume>2</volume><issue>12</issue><fpage>e0000397</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000397</pub-id><pub-id pub-id-type="medline">38039286</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yaneva</surname><given-names>V</given-names> </name><name name-style="western"><surname>Baldwin</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jurich</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Swygert</surname><given-names>K</given-names> </name><name name-style="western"><surname>Clauser</surname><given-names>BE</given-names> </name></person-group><article-title>Examining ChatGPT performance on USMLE sample items and implications for assessment</article-title><source>Acad Med</source><year>2024</year><month>02</month><day>1</day><volume>99</volume><issue>2</issue><fpage>192</fpage><lpage>197</lpage><pub-id pub-id-type="doi">10.1097/ACM.0000000000005549</pub-id><pub-id pub-id-type="medline">37934828</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pagano</surname><given-names>S</given-names> </name><name name-style="western"><surname>Strumolo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Michalk</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Evaluating ChatGPT, Gemini and other large language models (LLMs) in orthopaedic diagnostics: a prospective clinical study</article-title><source>Comput Struct Biotechnol J</source><year>2025</year><volume>28</volume><fpage>9</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1016/j.csbj.2024.12.013</pub-id><pub-id pub-id-type="medline">39850460</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Integrated image-based deep learning and language models for primary diabetes care</article-title><source>Nat Med</source><year>2024</year><month>10</month><volume>30</volume><issue>10</issue><fpage>2886</fpage><lpage>2896</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03139-8</pub-id><pub-id pub-id-type="medline">39030266</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ao</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Nie</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name></person-group><article-title>Comparative analysis of large language models on rare disease identification</article-title><source>Orphanet J Rare Dis</source><year>2025</year><month>04</month><day>1</day><volume>20</volume><issue>1</issue><fpage>150</fpage><pub-id pub-id-type="doi">10.1186/s13023-025-03656-w</pub-id><pub-id pub-id-type="medline">40165285</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rider</surname><given-names>NL</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chin</surname><given-names>AT</given-names> </name><etal/></person-group><article-title>Evaluating large language model performance to support the diagnosis and management of patients with primary immune disorders</article-title><source>J Allergy Clin Immunol</source><year>2025</year><month>07</month><volume>156</volume><issue>1</issue><fpage>81</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1016/j.jaci.2025.02.004</pub-id><pub-id pub-id-type="medline">39956279</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pillai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pillai</surname><given-names>K</given-names> </name></person-group><article-title>Accuracy of generative artificial intelligence models in differential diagnoses of familial Mediterranean fever and deficiency of Interleukin-1 receptor antagonist</article-title><source>J Transl Autoimmun</source><year>2023</year><month>12</month><volume>7</volume><fpage>100213</fpage><pub-id pub-id-type="doi">10.1016/j.jtauto.2023.100213</pub-id><pub-id pub-id-type="medline">37927888</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maharjan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Garikipati</surname><given-names>A</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>NP</given-names> </name><etal/></person-group><article-title>OpenMedLM: prompt engineering can out-perform fine-tuning in medical question-answering with open-source large language models</article-title><source>Sci Rep</source><year>2024</year><month>06</month><day>19</day><volume>14</volume><issue>1</issue><fpage>14156</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-64827-6</pub-id><pub-id pub-id-type="medline">38898116</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Z&#x00F6;ller</surname><given-names>N</given-names> </name><name name-style="western"><surname>Berger</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Human-AI collectives most accurately diagnose clinical vignettes</article-title><source>Proc Natl Acad Sci U S A</source><year>2025</year><month>06</month><day>17</day><volume>122</volume><issue>24</issue><fpage>e2426153122</fpage><pub-id pub-id-type="doi">10.1073/pnas.2426153122</pub-id><pub-id pub-id-type="medline">40512795</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Prevalence and risk factors of chronic obstructive pulmonary disease in China (the China pulmonary health [CPH] study): a national cross-sectional study</article-title><source>Lancet</source><year>2018</year><month>04</month><day>28</day><volume>391</volume><issue>10131</issue><fpage>1706</fpage><lpage>1717</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(18)30841-9</pub-id><pub-id pub-id-type="medline">29650248</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rhee</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Quinn</surname><given-names>KA</given-names> </name><etal/></person-group><article-title>Clinical manifestations and treatment in patients with relapsing polychondritis: a multicenter observational cohort study</article-title><source>ACR Open Rheumatol</source><year>2025</year><month>05</month><volume>7</volume><issue>5</issue><fpage>e70027</fpage><pub-id pub-id-type="doi">10.1002/acr2.70027</pub-id><pub-id pub-id-type="medline">40391876</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Rosas</surname><given-names>E</given-names> </name><name name-style="western"><surname>Asadigandomani</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Diagnostic performance of publicly available large language models in corneal diseases: a comparison with human specialists</article-title><source>Diagnostics (Basel)</source><year>2025</year><month>05</month><day>13</day><volume>15</volume><issue>10</issue><fpage>1221</fpage><pub-id pub-id-type="doi">10.3390/diagnostics15101221</pub-id><pub-id pub-id-type="medline">40428214</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abbasgholizadeh Rahimi</surname><given-names>S</given-names> </name><name name-style="western"><surname>L&#x00E9;gar&#x00E9;</surname><given-names>F</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Application of artificial intelligence in community-based primary health care: systematic scoping review and critical appraisal</article-title><source>J Med Internet Res</source><year>2021</year><month>09</month><day>3</day><volume>23</volume><issue>9</issue><fpage>e29839</fpage><pub-id pub-id-type="doi">10.2196/29839</pub-id><pub-id pub-id-type="medline">34477556</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>G&#x00FC;n</surname><given-names>M</given-names> </name></person-group><article-title>AI-assisted blood gas interpretation: a comparative study with an emergency physician</article-title><source>Am J Emerg Med</source><year>2025</year><month>08</month><volume>94</volume><fpage>1</fpage><lpage>2</lpage><pub-id pub-id-type="doi">10.1016/j.ajem.2025.04.028</pub-id><pub-id pub-id-type="medline">40252296</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rydzewski</surname><given-names>NR</given-names> </name><name name-style="western"><surname>Dinakaran</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>SG</given-names> </name><etal/></person-group><article-title>Comparative evaluation of LLMs in clinical oncology</article-title><source>NEJM AI</source><year>2024</year><month>05</month><volume>1</volume><issue>5</issue><fpage>39131700</fpage><pub-id pub-id-type="doi">10.1056/aioa2300151</pub-id><pub-id pub-id-type="medline">39131700</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Podlasek</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shidara</surname><given-names>K</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Alaa</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bernardo</surname><given-names>D</given-names> </name></person-group><article-title>Limitations of large language models in clinical problem-solving arising from inflexible reasoning</article-title><source>Sci Rep</source><year>2025</year><volume>15</volume><issue>1</issue><fpage>39426</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-22940-0</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>The application of large language models in medicine: a scoping review</article-title><source>iScience</source><year>2024</year><month>05</month><day>17</day><volume>27</volume><issue>5</issue><fpage>109713</fpage><pub-id pub-id-type="doi">10.1016/j.isci.2024.109713</pub-id><pub-id pub-id-type="medline">38746668</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Deep learning for rare disease: a scoping review</article-title><source>J Biomed Inform</source><year>2022</year><month>11</month><volume>135</volume><fpage>104227</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2022.104227</pub-id><pub-id pub-id-type="medline">36257483</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ling</surname><given-names>DI</given-names> </name><name name-style="western"><surname>Pai</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schiller</surname><given-names>I</given-names> </name><name name-style="western"><surname>Dendukuri</surname><given-names>N</given-names> </name></person-group><article-title>A Bayesian framework for estimating the incremental value of a diagnostic test in the absence of a gold standard</article-title><source>BMC Med Res Methodol</source><year>2014</year><month>05</month><day>15</day><volume>14</volume><fpage>67</fpage><pub-id pub-id-type="doi">10.1186/1471-2288-14-67</pub-id><pub-id pub-id-type="medline">24886359</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qiu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sha</surname><given-names>F</given-names> </name><name name-style="western"><surname>Allen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Linzen</surname><given-names>T</given-names> </name><name name-style="western"><surname>van Steenkiste</surname><given-names>S</given-names> </name></person-group><article-title>Bayesian teaching enables probabilistic reasoning in large language models</article-title><source>Nat Commun</source><year>2026</year><month>01</month><day>7</day><volume>17</volume><issue>1</issue><fpage>1238</fpage><pub-id pub-id-type="doi">10.1038/s41467-025-67998-6</pub-id><pub-id pub-id-type="medline">41501038</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qiu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Quantifying the reasoning abilities of LLMs on clinical cases</article-title><source>Nat Commun</source><year>2025</year><month>11</month><day>6</day><volume>16</volume><issue>1</issue><fpage>9799</fpage><pub-id pub-id-type="doi">10.1038/s41467-025-64769-1</pub-id><pub-id pub-id-type="medline">41198657</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Bommasani</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hudson</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Adeli</surname><given-names>E</given-names> </name><etal/></person-group><article-title>On the opportunities and risks of foundation models</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 16, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2108.07258</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yuan</surname><given-names>W</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>Y</given-names> </name></person-group><article-title>Large language models in cardiovascular imaging: current applications and future prospects</article-title><source>Med Research</source><year>2026</year><month>03</month><volume>2</volume><issue>1</issue><fpage>22</fpage><lpage>25</lpage><pub-id pub-id-type="doi">10.1002/mdr2.70042</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Evaluation framework and diagnostic performance analysis of large language models.</p><media xlink:href="jmir_v28i1e89963_app1.doc" xlink:title="DOC File, 2166 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Sensitivity analysis results for generalized estimating equation (GEE) models.</p><media xlink:href="jmir_v28i1e89963_app2.xlsx" xlink:title="XLSX File, 20 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>STARD 2015-Checklist.</p><media xlink:href="jmir_v28i1e89963_app3.doc" xlink:title="DOC File, 88 KB"/></supplementary-material></app-group></back></article>